Spaces:
Sleeping
Sleeping
Invalid JSON:Unexpected token 'N', ..."ad_norm": NaN,
"... is not valid JSON
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 3.0, | |
| "eval_steps": 500, | |
| "global_step": 384, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 23.625, | |
| "completions/mean_terminated_length": 23.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09965698793530464, | |
| "epoch": 0.0078125, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.13743257522583, | |
| "kl": 9.029518110992285e-08, | |
| "learning_rate": 0.0, | |
| "loss": -0.06901008635759354, | |
| "num_tokens": 4097.0, | |
| "reward": 0.768750011920929, | |
| "reward_std": 0.4300643801689148, | |
| "rewards/reward_fn/mean": 0.768750011920929, | |
| "rewards/reward_fn/std": 0.4300643801689148, | |
| "step": 1, | |
| "step_time": 7.910418492999952 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 19.625, | |
| "completions/mean_terminated_length": 19.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13235441222786903, | |
| "epoch": 0.015625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0, | |
| "kl": 0.0, | |
| "learning_rate": 1.282051282051282e-07, | |
| "loss": 0.0, | |
| "num_tokens": 9674.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 2, | |
| "step_time": 7.016720726000017 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 22.5, | |
| "completions/mean_terminated_length": 22.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11623429134488106, | |
| "epoch": 0.0234375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0009358773822896183, | |
| "kl": 8.32239061310247e-06, | |
| "learning_rate": 2.564102564102564e-07, | |
| "loss": 8.256898098579768e-08, | |
| "num_tokens": 14558.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 3, | |
| "step_time": 6.771791392000068 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 23.5, | |
| "completions/mean_terminated_length": 23.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09514202550053596, | |
| "epoch": 0.03125, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.5667312145233154, | |
| "kl": 3.819915718850098e-06, | |
| "learning_rate": 3.846153846153847e-07, | |
| "loss": -0.08508844673633575, | |
| "num_tokens": 18626.0, | |
| "reward": 0.8812500238418579, | |
| "reward_std": 0.3358757197856903, | |
| "rewards/reward_fn/mean": 0.8812500238418579, | |
| "rewards/reward_fn/std": 0.3358757197856903, | |
| "step": 4, | |
| "step_time": 5.745465063999859 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.14030911028385162, | |
| "epoch": 0.0390625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0005072976928204298, | |
| "kl": 4.893604227618198e-06, | |
| "learning_rate": 5.128205128205128e-07, | |
| "loss": 4.175808498985134e-08, | |
| "num_tokens": 24180.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 5, | |
| "step_time": 7.145598770999982 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.375, | |
| "completions/mean_terminated_length": 21.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11003290489315987, | |
| "epoch": 0.046875, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 3.21610426902771, | |
| "kl": 1.439371067135653e-05, | |
| "learning_rate": 6.41025641025641e-07, | |
| "loss": 0.008770093321800232, | |
| "num_tokens": 28935.0, | |
| "reward": 0.887499988079071, | |
| "reward_std": 0.3181980550289154, | |
| "rewards/reward_fn/mean": 0.887499988079071, | |
| "rewards/reward_fn/std": 0.3181980550289154, | |
| "step": 6, | |
| "step_time": 6.531211202000009 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.625, | |
| "completions/mean_terminated_length": 21.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12766291946172714, | |
| "epoch": 0.0546875, | |
| "frac_reward_zero_std": 0.0, | |
| "grad_norm": 2.585819959640503, | |
| "kl": 2.266431465614005e-06, | |
| "learning_rate": 7.692307692307694e-07, | |
| "loss": -0.25185173749923706, | |
| "num_tokens": 32988.0, | |
| "reward": 0.643750011920929, | |
| "reward_std": 0.49384605884552, | |
| "rewards/reward_fn/mean": 0.643750011920929, | |
| "rewards/reward_fn/std": 0.49384605884552, | |
| "step": 7, | |
| "step_time": 5.520926363000058 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.75, | |
| "completions/mean_terminated_length": 21.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11385262385010719, | |
| "epoch": 0.0625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00018979530432261527, | |
| "kl": 3.1351814868685324e-06, | |
| "learning_rate": 8.974358974358975e-07, | |
| "loss": 3.0823137819879776e-08, | |
| "num_tokens": 37866.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 8, | |
| "step_time": 6.259193151999966 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.25, | |
| "completions/mean_terminated_length": 13.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1592414677143097, | |
| "epoch": 0.0703125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0010657254606485367, | |
| "kl": 1.2110989018765395e-05, | |
| "learning_rate": 1.0256410256410257e-06, | |
| "loss": 1.211098918929565e-07, | |
| "num_tokens": 43344.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 9, | |
| "step_time": 4.96420772700003 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1171259917318821, | |
| "epoch": 0.078125, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.561084508895874, | |
| "kl": 7.743469495835598e-06, | |
| "learning_rate": 1.153846153846154e-06, | |
| "loss": -0.05262097343802452, | |
| "num_tokens": 47400.0, | |
| "reward": 0.875, | |
| "reward_std": 0.3535533845424652, | |
| "rewards/reward_fn/mean": 0.875, | |
| "rewards/reward_fn/std": 0.3535533845424652, | |
| "step": 10, | |
| "step_time": 5.405768857999988 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 13.0, | |
| "completions/max_terminated_length": 13.0, | |
| "completions/mean_length": 13.0, | |
| "completions/mean_terminated_length": 13.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.16070035845041275, | |
| "epoch": 0.0859375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00039809694862924516, | |
| "kl": 4.659478577195841e-06, | |
| "learning_rate": 1.282051282051282e-06, | |
| "loss": 4.659478491930713e-08, | |
| "num_tokens": 52904.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 11, | |
| "step_time": 5.233364768000001 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 23.25, | |
| "completions/mean_terminated_length": 23.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11492524296045303, | |
| "epoch": 0.09375, | |
| "frac_reward_zero_std": 0.0, | |
| "grad_norm": NaN, | |
| "kl": 8.350619191332953e-05, | |
| "learning_rate": 1.4102564102564104e-06, | |
| "loss": -0.1988813430070877, | |
| "num_tokens": 57022.0, | |
| "reward": 0.7687499523162842, | |
| "reward_std": 0.4284002482891083, | |
| "rewards/reward_fn/mean": 0.7687499523162842, | |
| "rewards/reward_fn/std": 0.4284002482891083, | |
| "step": 12, | |
| "step_time": 5.546596589000046 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.125, | |
| "completions/mean_terminated_length": 19.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.17772966623306274, | |
| "epoch": 0.1015625, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.7892787456512451, | |
| "kl": 1.1041468042094493e-05, | |
| "learning_rate": 1.5384615384615387e-06, | |
| "loss": -0.147029310464859, | |
| "num_tokens": 61727.0, | |
| "reward": 0.875, | |
| "reward_std": 0.3535533845424652, | |
| "rewards/reward_fn/mean": 0.875, | |
| "rewards/reward_fn/std": 0.3535533845424652, | |
| "step": 13, | |
| "step_time": 6.295519733999981 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.0, | |
| "completions/mean_terminated_length": 21.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13657044991850853, | |
| "epoch": 0.109375, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 2.4697511196136475, | |
| "kl": 0.00011959002404182684, | |
| "learning_rate": 1.6666666666666667e-06, | |
| "loss": -0.07215305417776108, | |
| "num_tokens": 65771.0, | |
| "reward": 0.7875000238418579, | |
| "reward_std": 0.3934735357761383, | |
| "rewards/reward_fn/mean": 0.7875000238418579, | |
| "rewards/reward_fn/std": 0.3934735655784607, | |
| "step": 14, | |
| "step_time": 5.657076707999977 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 66.0, | |
| "completions/max_terminated_length": 66.0, | |
| "completions/mean_length": 29.0, | |
| "completions/mean_terminated_length": 29.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.10727399960160255, | |
| "epoch": 0.1171875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0006498436559922993, | |
| "kl": 2.7017442334908992e-05, | |
| "learning_rate": 1.794871794871795e-06, | |
| "loss": 2.6791002483150805e-07, | |
| "num_tokens": 70595.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 15, | |
| "step_time": 9.642551118000029 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 22.0, | |
| "completions/mean_terminated_length": 22.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09909634664654732, | |
| "epoch": 0.125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0024827050510793924, | |
| "kl": 6.381553976098076e-05, | |
| "learning_rate": 1.9230769230769234e-06, | |
| "loss": 6.585330538655398e-07, | |
| "num_tokens": 75331.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 16, | |
| "step_time": 7.090756369000019 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.625, | |
| "completions/mean_terminated_length": 19.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11970487236976624, | |
| "epoch": 0.1328125, | |
| "frac_reward_zero_std": 0.0, | |
| "grad_norm": 2.5904784202575684, | |
| "kl": 0.0007240207705763169, | |
| "learning_rate": 2.0512820512820513e-06, | |
| "loss": -0.2588059902191162, | |
| "num_tokens": 79512.0, | |
| "reward": 0.5437500476837158, | |
| "reward_std": 0.49021676182746887, | |
| "rewards/reward_fn/mean": 0.5437500476837158, | |
| "rewards/reward_fn/std": 0.49021679162979126, | |
| "step": 17, | |
| "step_time": 5.971762764999994 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.125, | |
| "completions/mean_terminated_length": 19.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.10251206904649734, | |
| "epoch": 0.140625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.003479904029518366, | |
| "kl": 0.00019486098608467728, | |
| "learning_rate": 2.1794871794871797e-06, | |
| "loss": 1.9753097149077803e-06, | |
| "num_tokens": 84289.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 18, | |
| "step_time": 6.534459332000097 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 40.0, | |
| "completions/max_terminated_length": 40.0, | |
| "completions/mean_length": 16.875, | |
| "completions/mean_terminated_length": 16.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13044045120477676, | |
| "epoch": 0.1484375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006375588476657867, | |
| "kl": 0.00021761858806712553, | |
| "learning_rate": 2.307692307692308e-06, | |
| "loss": 2.2181316126079764e-06, | |
| "num_tokens": 89848.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 19, | |
| "step_time": 8.074243104000061 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 25.125, | |
| "completions/mean_terminated_length": 25.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07144641503691673, | |
| "epoch": 0.15625, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.383427381515503, | |
| "kl": 0.00042095681419596076, | |
| "learning_rate": 2.435897435897436e-06, | |
| "loss": -0.111912801861763, | |
| "num_tokens": 93977.0, | |
| "reward": 0.8812500238418579, | |
| "reward_std": 0.3358757197856903, | |
| "rewards/reward_fn/mean": 0.8812500238418579, | |
| "rewards/reward_fn/std": 0.3358757197856903, | |
| "step": 20, | |
| "step_time": 5.788662745999886 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.375, | |
| "completions/mean_terminated_length": 19.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.10241465270519257, | |
| "epoch": 0.1640625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0049729375168681145, | |
| "kl": 0.00047132828331086785, | |
| "learning_rate": 2.564102564102564e-06, | |
| "loss": 4.8525371312280186e-06, | |
| "num_tokens": 98852.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 21, | |
| "step_time": 6.948417052999957 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 24.75, | |
| "completions/mean_terminated_length": 24.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09643011912703514, | |
| "epoch": 0.171875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006859563756734133, | |
| "kl": 0.0007727623451501131, | |
| "learning_rate": 2.6923076923076923e-06, | |
| "loss": 6.733347163390135e-06, | |
| "num_tokens": 103734.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 22, | |
| "step_time": 7.359714186000019 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 47.0, | |
| "completions/max_terminated_length": 47.0, | |
| "completions/mean_length": 23.375, | |
| "completions/mean_terminated_length": 23.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.42096097022295, | |
| "epoch": 0.1796875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005077087786048651, | |
| "kl": 0.0007580214296467602, | |
| "learning_rate": 2.8205128205128207e-06, | |
| "loss": 7.698860827076714e-06, | |
| "num_tokens": 108625.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 23, | |
| "step_time": 8.31426027000009 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 30.0, | |
| "completions/mean_terminated_length": 30.0, | |
| "completions/min_length": 30.0, | |
| "completions/min_terminated_length": 30.0, | |
| "entropy": 0.05439547263085842, | |
| "epoch": 0.1875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.000995291629806161, | |
| "kl": 0.0004853812279179692, | |
| "learning_rate": 2.948717948717949e-06, | |
| "loss": 4.853812242799904e-06, | |
| "num_tokens": 112781.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 24, | |
| "step_time": 5.9903554890000805 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 37.0, | |
| "completions/max_terminated_length": 37.0, | |
| "completions/mean_length": 24.375, | |
| "completions/mean_terminated_length": 24.375, | |
| "completions/min_length": 14.0, | |
| "completions/min_terminated_length": 14.0, | |
| "entropy": 0.10691225156188011, | |
| "epoch": 0.1953125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.011116056703031063, | |
| "kl": 0.0012052947713527828, | |
| "learning_rate": 3.0769230769230774e-06, | |
| "loss": 1.1946971426368691e-05, | |
| "num_tokens": 117612.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 25, | |
| "step_time": 7.288805328999956 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 25.0, | |
| "completions/mean_terminated_length": 25.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1299378089606762, | |
| "epoch": 0.203125, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.851197361946106, | |
| "kl": 0.002050661132670939, | |
| "learning_rate": 3.205128205128206e-06, | |
| "loss": -0.11245696991682053, | |
| "num_tokens": 122448.0, | |
| "reward": 0.875, | |
| "reward_std": 0.3535533845424652, | |
| "rewards/reward_fn/mean": 0.875, | |
| "rewards/reward_fn/std": 0.3535533845424652, | |
| "step": 26, | |
| "step_time": 7.037999247000016 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.375, | |
| "completions/mean_terminated_length": 21.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07765422388911247, | |
| "epoch": 0.2109375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00900979246944189, | |
| "kl": 0.001701147179119289, | |
| "learning_rate": 3.3333333333333333e-06, | |
| "loss": 1.7008303984766826e-05, | |
| "num_tokens": 127211.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 27, | |
| "step_time": 6.768718518000014 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 28.0, | |
| "completions/mean_terminated_length": 28.0, | |
| "completions/min_length": 14.0, | |
| "completions/min_terminated_length": 14.0, | |
| "entropy": 0.053295012563467026, | |
| "epoch": 0.21875, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.0311310291290283, | |
| "kl": 0.005389484780607745, | |
| "learning_rate": 3.4615384615384617e-06, | |
| "loss": -0.10706700384616852, | |
| "num_tokens": 131351.0, | |
| "reward": 0.893750011920929, | |
| "reward_std": 0.3005203604698181, | |
| "rewards/reward_fn/mean": 0.893750011920929, | |
| "rewards/reward_fn/std": 0.3005203604698181, | |
| "step": 28, | |
| "step_time": 5.859235952000063 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 18.5, | |
| "completions/mean_terminated_length": 18.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08729634806513786, | |
| "epoch": 0.2265625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.11862029135227203, | |
| "kl": 0.00995962810702622, | |
| "learning_rate": 3.58974358974359e-06, | |
| "loss": 0.00010122068488271907, | |
| "num_tokens": 136067.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 29, | |
| "step_time": 7.392926845999909 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 25.125, | |
| "completions/mean_terminated_length": 25.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.049397675320506096, | |
| "epoch": 0.234375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.013916196301579475, | |
| "kl": 0.004311037337174639, | |
| "learning_rate": 3.7179487179487184e-06, | |
| "loss": 3.796327655436471e-05, | |
| "num_tokens": 140216.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 30, | |
| "step_time": 5.707905451999977 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.125, | |
| "completions/max_length": 128.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 44.625, | |
| "completions/mean_terminated_length": 32.71428680419922, | |
| "completions/min_length": 14.0, | |
| "completions/min_terminated_length": 14.0, | |
| "entropy": 0.3870660662651062, | |
| "epoch": 0.2421875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00857287272810936, | |
| "kl": 0.0027850136975757778, | |
| "learning_rate": 3.846153846153847e-06, | |
| "loss": 2.3635355319129303e-05, | |
| "num_tokens": 145949.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 31, | |
| "step_time": 14.826277505999997 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 29.25, | |
| "completions/mean_terminated_length": 29.25, | |
| "completions/min_length": 14.0, | |
| "completions/min_terminated_length": 14.0, | |
| "entropy": 0.15847552567720413, | |
| "epoch": 0.25, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 3.0680534839630127, | |
| "kl": 0.005178321152925491, | |
| "learning_rate": 3.974358974358974e-06, | |
| "loss": 0.021413102746009827, | |
| "num_tokens": 150875.0, | |
| "reward": 0.875, | |
| "reward_std": 0.3535533845424652, | |
| "rewards/reward_fn/mean": 0.875, | |
| "rewards/reward_fn/std": 0.3535533845424652, | |
| "step": 32, | |
| "step_time": 7.193657256999927 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.375, | |
| "completions/mean_terminated_length": 27.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.036486370489001274, | |
| "epoch": 0.2578125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.013429846614599228, | |
| "kl": 0.0038138862000778317, | |
| "learning_rate": 4.102564102564103e-06, | |
| "loss": 3.6106022889725864e-05, | |
| "num_tokens": 155090.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 33, | |
| "step_time": 5.989709379999908 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 24.75, | |
| "completions/mean_terminated_length": 24.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06528343632817268, | |
| "epoch": 0.265625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.23420189321041107, | |
| "kl": 0.015011879149824381, | |
| "learning_rate": 4.230769230769231e-06, | |
| "loss": 0.00015618561883457005, | |
| "num_tokens": 159860.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 34, | |
| "step_time": 7.343351643999995 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 24.75, | |
| "completions/mean_terminated_length": 24.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09101804345846176, | |
| "epoch": 0.2734375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0637916773557663, | |
| "kl": 0.011130816768854856, | |
| "learning_rate": 4.358974358974359e-06, | |
| "loss": 0.00011102524877060205, | |
| "num_tokens": 165430.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 35, | |
| "step_time": 7.710919977999993 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 27.0, | |
| "completions/mean_terminated_length": 27.0, | |
| "completions/min_length": 14.0, | |
| "completions/min_terminated_length": 14.0, | |
| "entropy": 0.06510375998914242, | |
| "epoch": 0.28125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03826780989766121, | |
| "kl": 0.0053558857180178165, | |
| "learning_rate": 4.487179487179488e-06, | |
| "loss": 5.355885878088884e-05, | |
| "num_tokens": 170254.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 36, | |
| "step_time": 7.2804435799999965 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 27.125, | |
| "completions/mean_terminated_length": 27.125, | |
| "completions/min_length": 14.0, | |
| "completions/min_terminated_length": 14.0, | |
| "entropy": 0.043790558353066444, | |
| "epoch": 0.2890625, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.402679681777954, | |
| "kl": 0.02490220731124282, | |
| "learning_rate": 4.615384615384616e-06, | |
| "loss": 0.03481510281562805, | |
| "num_tokens": 174395.0, | |
| "reward": 0.875, | |
| "reward_std": 0.3535533845424652, | |
| "rewards/reward_fn/mean": 0.875, | |
| "rewards/reward_fn/std": 0.3535533845424652, | |
| "step": 37, | |
| "step_time": 5.756117628000084 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 18.875, | |
| "completions/mean_terminated_length": 18.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12046026065945625, | |
| "epoch": 0.296875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.054251085966825485, | |
| "kl": 0.018674671184271574, | |
| "learning_rate": 4.743589743589744e-06, | |
| "loss": 0.00016035939916037023, | |
| "num_tokens": 179942.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 38, | |
| "step_time": 7.3252331490000415 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 26.375, | |
| "completions/mean_terminated_length": 26.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05612089857459068, | |
| "epoch": 0.3046875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.039188120514154434, | |
| "kl": 0.012932237819768488, | |
| "learning_rate": 4.871794871794872e-06, | |
| "loss": 0.00010084287350764498, | |
| "num_tokens": 184741.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 39, | |
| "step_time": 7.515498236999974 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 24.75, | |
| "completions/mean_terminated_length": 24.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09634075127542019, | |
| "epoch": 0.3125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.018983978778123856, | |
| "kl": 0.005221477011218667, | |
| "learning_rate": 5e-06, | |
| "loss": 5.3893018048256636e-05, | |
| "num_tokens": 189627.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 40, | |
| "step_time": 7.176626092999982 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.25, | |
| "completions/mean_terminated_length": 21.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06109755299985409, | |
| "epoch": 0.3203125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.030976427718997, | |
| "kl": 0.009227571543306112, | |
| "learning_rate": 4.999896350176413e-06, | |
| "loss": 9.208174014929682e-05, | |
| "num_tokens": 194497.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 41, | |
| "step_time": 6.542497393999952 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 20.625, | |
| "completions/mean_terminated_length": 20.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.17987589165568352, | |
| "epoch": 0.328125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0431177020072937, | |
| "kl": 0.015479911118745804, | |
| "learning_rate": 4.999585409300281e-06, | |
| "loss": 0.0001403249625582248, | |
| "num_tokens": 200082.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 42, | |
| "step_time": 7.701265004999868 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 25.125, | |
| "completions/mean_terminated_length": 25.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07510707713663578, | |
| "epoch": 0.3359375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03998936340212822, | |
| "kl": 0.012762806843966246, | |
| "learning_rate": 4.999067203154777e-06, | |
| "loss": 0.00010049781849374995, | |
| "num_tokens": 204927.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 43, | |
| "step_time": 7.225048154999968 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.5, | |
| "completions/mean_terminated_length": 19.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06112176366150379, | |
| "epoch": 0.34375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.028579436242580414, | |
| "kl": 0.01107706013135612, | |
| "learning_rate": 4.998341774709482e-06, | |
| "loss": 0.00010365620255470276, | |
| "num_tokens": 209771.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 44, | |
| "step_time": 6.643636920000063 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 44.0, | |
| "completions/max_terminated_length": 44.0, | |
| "completions/mean_length": 27.875, | |
| "completions/mean_terminated_length": 27.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12446310371160507, | |
| "epoch": 0.3515625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.012319399043917656, | |
| "kl": 0.004381780279800296, | |
| "learning_rate": 4.9974091841168195e-06, | |
| "loss": 4.3796011595986784e-05, | |
| "num_tokens": 214682.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 45, | |
| "step_time": 7.909422166999889 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 22.375, | |
| "completions/mean_terminated_length": 22.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.10363675281405449, | |
| "epoch": 0.359375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.04150591790676117, | |
| "kl": 0.007062526885420084, | |
| "learning_rate": 4.99626950870707e-06, | |
| "loss": 6.894840043969452e-05, | |
| "num_tokens": 220237.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 46, | |
| "step_time": 7.691492860000039 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.5, | |
| "completions/mean_terminated_length": 27.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.030814163386821747, | |
| "epoch": 0.3671875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.013106024824082851, | |
| "kl": 0.0039410877507179976, | |
| "learning_rate": 4.994922842981958e-06, | |
| "loss": 3.779985854635015e-05, | |
| "num_tokens": 224417.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 47, | |
| "step_time": 5.999039098000026 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 16.625, | |
| "completions/mean_terminated_length": 16.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1171766109764576, | |
| "epoch": 0.375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.04344555363059044, | |
| "kl": 0.010649031959474087, | |
| "learning_rate": 4.993369298606817e-06, | |
| "loss": 0.00010028588440036401, | |
| "num_tokens": 229926.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 48, | |
| "step_time": 7.581720861000008 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.625, | |
| "completions/mean_terminated_length": 13.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1510508954524994, | |
| "epoch": 0.3828125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.04870390519499779, | |
| "kl": 0.011529957875609398, | |
| "learning_rate": 4.991609004401324e-06, | |
| "loss": 0.0001149907911894843, | |
| "num_tokens": 235435.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 49, | |
| "step_time": 5.576136509999969 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.0, | |
| "completions/mean_terminated_length": 17.0, | |
| "completions/min_length": 12.0, | |
| "completions/min_terminated_length": 12.0, | |
| "entropy": 0.13440731912851334, | |
| "epoch": 0.390625, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.8115813732147217, | |
| "kl": 0.20483593340031803, | |
| "learning_rate": 4.989642106328829e-06, | |
| "loss": -0.1266314685344696, | |
| "num_tokens": 240183.0, | |
| "reward": 0.875, | |
| "reward_std": 0.3535533845424652, | |
| "rewards/reward_fn/mean": 0.875, | |
| "rewards/reward_fn/std": 0.3535533845424652, | |
| "step": 50, | |
| "step_time": 6.442843831000005 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.375, | |
| "completions/mean_terminated_length": 27.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.029575850814580917, | |
| "epoch": 0.3984375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006918889936059713, | |
| "kl": 0.0022675381042063236, | |
| "learning_rate": 4.98746876748424e-06, | |
| "loss": 2.1996882423991337e-05, | |
| "num_tokens": 244150.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 51, | |
| "step_time": 5.648842897999998 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.75, | |
| "completions/mean_terminated_length": 13.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12618596106767654, | |
| "epoch": 0.40625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.04183642938733101, | |
| "kl": 0.007541614351794124, | |
| "learning_rate": 4.985089168080509e-06, | |
| "loss": 7.499127241317183e-05, | |
| "num_tokens": 249636.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 52, | |
| "step_time": 5.985882786999923 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 27.0, | |
| "completions/mean_terminated_length": 27.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.027834360487759113, | |
| "epoch": 0.4140625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005657592322677374, | |
| "kl": 0.0021601368789561093, | |
| "learning_rate": 4.982503505433683e-06, | |
| "loss": 2.106852480210364e-05, | |
| "num_tokens": 253604.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 53, | |
| "step_time": 5.815157560000102 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 49.0, | |
| "completions/max_terminated_length": 49.0, | |
| "completions/mean_length": 30.0, | |
| "completions/mean_terminated_length": 30.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.049247266724705696, | |
| "epoch": 0.421875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008218275383114815, | |
| "kl": 0.0025316422106698155, | |
| "learning_rate": 4.979711993946543e-06, | |
| "loss": 2.4405037038377486e-05, | |
| "num_tokens": 257780.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 54, | |
| "step_time": 7.388005351000061 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.125, | |
| "completions/max_length": 128.0, | |
| "completions/max_terminated_length": 71.0, | |
| "completions/mean_length": 38.25, | |
| "completions/mean_terminated_length": 25.428571701049805, | |
| "completions/min_length": 14.0, | |
| "completions/min_terminated_length": 14.0, | |
| "entropy": 0.14161667972803116, | |
| "epoch": 0.4296875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.011416819877922535, | |
| "kl": 0.0019309421186335385, | |
| "learning_rate": 4.976714865090827e-06, | |
| "loss": 2.0552859496092424e-05, | |
| "num_tokens": 263510.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 55, | |
| "step_time": 15.045006353999952 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 19.125, | |
| "completions/mean_terminated_length": 19.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11700041592121124, | |
| "epoch": 0.4375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01164371706545353, | |
| "kl": 0.001374046492855996, | |
| "learning_rate": 4.973512367388038e-06, | |
| "loss": 1.2746955690090545e-05, | |
| "num_tokens": 269087.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 56, | |
| "step_time": 7.920732168999962 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 94.0, | |
| "completions/max_terminated_length": 94.0, | |
| "completions/mean_length": 37.625, | |
| "completions/mean_terminated_length": 37.625, | |
| "completions/min_length": 29.0, | |
| "completions/min_terminated_length": 29.0, | |
| "entropy": 0.04022482968866825, | |
| "epoch": 0.4453125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0068351044319570065, | |
| "kl": 0.0017894834163598716, | |
| "learning_rate": 4.970104766388833e-06, | |
| "loss": 1.8479673599358648e-05, | |
| "num_tokens": 273396.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 57, | |
| "step_time": 11.183577817000014 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 23.25, | |
| "completions/mean_terminated_length": 23.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06758509948849678, | |
| "epoch": 0.453125, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.572488784790039, | |
| "kl": 0.0030532198143191636, | |
| "learning_rate": 4.966492344651006e-06, | |
| "loss": -0.07791005074977875, | |
| "num_tokens": 278234.0, | |
| "reward": 0.8812500238418579, | |
| "reward_std": 0.3358757197856903, | |
| "rewards/reward_fn/mean": 0.8812500238418579, | |
| "rewards/reward_fn/std": 0.3358757197856903, | |
| "step": 58, | |
| "step_time": 7.65875285900006 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 27.0, | |
| "completions/mean_terminated_length": 27.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.033227477222681046, | |
| "epoch": 0.4609375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0016361080342903733, | |
| "kl": 0.0016750092036090791, | |
| "learning_rate": 4.962675401716056e-06, | |
| "loss": 1.6619302186882123e-05, | |
| "num_tokens": 282454.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 59, | |
| "step_time": 5.975613750999969 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.375, | |
| "completions/mean_terminated_length": 13.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13986211270093918, | |
| "epoch": 0.46875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.04219949617981911, | |
| "kl": 0.004549495875835419, | |
| "learning_rate": 4.958654254084356e-06, | |
| "loss": 4.563133552437648e-05, | |
| "num_tokens": 287961.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 60, | |
| "step_time": 5.713278319999972 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.25, | |
| "completions/mean_terminated_length": 21.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0659055095165968, | |
| "epoch": 0.4765625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.010506045073270798, | |
| "kl": 0.0030614889692515135, | |
| "learning_rate": 4.954429235188897e-06, | |
| "loss": 2.8885815481771715e-05, | |
| "num_tokens": 292807.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 61, | |
| "step_time": 6.593735059999972 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 25.375, | |
| "completions/mean_terminated_length": 25.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.040277909487485886, | |
| "epoch": 0.484375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006953485310077667, | |
| "kl": 0.0017460978706367314, | |
| "learning_rate": 4.95000069536765e-06, | |
| "loss": 1.6949392374954186e-05, | |
| "num_tokens": 296982.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 62, | |
| "step_time": 5.818303646000004 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.375, | |
| "completions/mean_terminated_length": 13.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.14363406598567963, | |
| "epoch": 0.4921875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0677562728524208, | |
| "kl": 0.00942042376846075, | |
| "learning_rate": 4.9453690018345144e-06, | |
| "loss": 9.429677447769791e-05, | |
| "num_tokens": 302489.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 63, | |
| "step_time": 5.593983074999983 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.25, | |
| "completions/mean_terminated_length": 13.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12100231274962425, | |
| "epoch": 0.5, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01809072121977806, | |
| "kl": 0.003618443734012544, | |
| "learning_rate": 4.940534538648862e-06, | |
| "loss": 3.6511512007564306e-05, | |
| "num_tokens": 308019.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 64, | |
| "step_time": 5.77941624999994 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.5, | |
| "completions/mean_terminated_length": 27.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03561065532267094, | |
| "epoch": 0.5078125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0022499635815620422, | |
| "kl": 0.0012880651047453284, | |
| "learning_rate": 4.935497706683698e-06, | |
| "loss": 1.28087995108217e-05, | |
| "num_tokens": 312083.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 65, | |
| "step_time": 5.752014847000055 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 15.25, | |
| "completions/mean_terminated_length": 15.25, | |
| "completions/min_length": 12.0, | |
| "completions/min_terminated_length": 12.0, | |
| "entropy": 0.14075982943177223, | |
| "epoch": 0.515625, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 2.6456313133239746, | |
| "kl": 0.09695499250665307, | |
| "learning_rate": 4.9302589235924185e-06, | |
| "loss": -0.08113337308168411, | |
| "num_tokens": 316841.0, | |
| "reward": 0.875, | |
| "reward_std": 0.3535533845424652, | |
| "rewards/reward_fn/mean": 0.875, | |
| "rewards/reward_fn/std": 0.3535533845424652, | |
| "step": 66, | |
| "step_time": 6.563686877999999 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.125, | |
| "completions/mean_terminated_length": 19.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07007651403546333, | |
| "epoch": 0.5234375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.017309362068772316, | |
| "kl": 0.004948096117004752, | |
| "learning_rate": 4.924818623774178e-06, | |
| "loss": 4.838653694605455e-05, | |
| "num_tokens": 321630.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 67, | |
| "step_time": 6.570541892999927 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 23.375, | |
| "completions/mean_terminated_length": 23.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0408208854496479, | |
| "epoch": 0.53125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.012536097317934036, | |
| "kl": 0.004252667655237019, | |
| "learning_rate": 4.91917725833787e-06, | |
| "loss": 3.6156445275992155e-05, | |
| "num_tokens": 325569.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 68, | |
| "step_time": 5.749754669000026 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 13.0, | |
| "completions/max_terminated_length": 13.0, | |
| "completions/mean_length": 13.0, | |
| "completions/mean_terminated_length": 13.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1359182596206665, | |
| "epoch": 0.5390625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03752928227186203, | |
| "kl": 0.009188917931169271, | |
| "learning_rate": 4.913335295064721e-06, | |
| "loss": 9.188917465507984e-05, | |
| "num_tokens": 331045.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 69, | |
| "step_time": 5.468208492999906 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.5, | |
| "completions/mean_terminated_length": 19.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07358374819159508, | |
| "epoch": 0.546875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.020383596420288086, | |
| "kl": 0.005584244150668383, | |
| "learning_rate": 4.907293218369499e-06, | |
| "loss": 5.529334521270357e-05, | |
| "num_tokens": 335761.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 70, | |
| "step_time": 6.64621212499992 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.0, | |
| "completions/mean_terminated_length": 17.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05005803145468235, | |
| "epoch": 0.5546875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03125562518835068, | |
| "kl": 0.008841587696224451, | |
| "learning_rate": 4.901051529260352e-06, | |
| "loss": 8.841587987262756e-05, | |
| "num_tokens": 339757.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 71, | |
| "step_time": 5.899700159999952 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 17.25, | |
| "completions/mean_terminated_length": 17.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07237722724676132, | |
| "epoch": 0.5625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.018481554463505745, | |
| "kl": 0.006117145298048854, | |
| "learning_rate": 4.89461074529726e-06, | |
| "loss": 6.117144948802888e-05, | |
| "num_tokens": 344555.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 72, | |
| "step_time": 6.860756025000001 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.0, | |
| "completions/mean_terminated_length": 21.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.045137686654925346, | |
| "epoch": 0.5703125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.020073302090168, | |
| "kl": 0.005907518556341529, | |
| "learning_rate": 4.8879714005491205e-06, | |
| "loss": 5.907518061576411e-05, | |
| "num_tokens": 348591.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 73, | |
| "step_time": 5.62259969999991 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.125, | |
| "completions/mean_terminated_length": 19.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0720517449080944, | |
| "epoch": 0.578125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.027030251920223236, | |
| "kl": 0.01036432757973671, | |
| "learning_rate": 4.881134045549463e-06, | |
| "loss": 8.4122279076837e-05, | |
| "num_tokens": 353312.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 74, | |
| "step_time": 6.612577034999958 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.125, | |
| "completions/max_length": 128.0, | |
| "completions/max_terminated_length": 13.0, | |
| "completions/mean_length": 27.375, | |
| "completions/mean_terminated_length": 13.000000953674316, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1056247390806675, | |
| "epoch": 0.5859375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.019467800855636597, | |
| "kl": 0.006749335676431656, | |
| "learning_rate": 4.874099247250799e-06, | |
| "loss": 5.878092997591011e-05, | |
| "num_tokens": 358931.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 75, | |
| "step_time": 14.855594667999867 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 17.125, | |
| "completions/mean_terminated_length": 17.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.048843516036868095, | |
| "epoch": 0.59375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.033446282148361206, | |
| "kl": 0.012164841871708632, | |
| "learning_rate": 4.8668675889776095e-06, | |
| "loss": 0.00012165858061052859, | |
| "num_tokens": 362844.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 76, | |
| "step_time": 5.699464158999945 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.0, | |
| "completions/mean_terminated_length": 17.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07759269699454308, | |
| "epoch": 0.6015625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.017857957631349564, | |
| "kl": 0.0074398701544851065, | |
| "learning_rate": 4.85943967037798e-06, | |
| "loss": 6.409453635569662e-05, | |
| "num_tokens": 367560.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 77, | |
| "step_time": 6.886273725999672 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 15.875, | |
| "completions/mean_terminated_length": 15.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12127615511417389, | |
| "epoch": 0.609375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.024566544219851494, | |
| "kl": 0.007913533365353942, | |
| "learning_rate": 4.851816107373871e-06, | |
| "loss": 7.854383147787303e-05, | |
| "num_tokens": 373083.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 78, | |
| "step_time": 7.353863909999973 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.25, | |
| "completions/mean_terminated_length": 13.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13099069893360138, | |
| "epoch": 0.6171875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0477265790104866, | |
| "kl": 0.011370216961950064, | |
| "learning_rate": 4.843997532110051e-06, | |
| "loss": 0.00011370217544026673, | |
| "num_tokens": 378589.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 79, | |
| "step_time": 5.622442682000155 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07205060124397278, | |
| "epoch": 0.625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.012449050322175026, | |
| "kl": 0.00409984530415386, | |
| "learning_rate": 4.835984592901678e-06, | |
| "loss": 3.9597347495146096e-05, | |
| "num_tokens": 383357.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 80, | |
| "step_time": 6.542991357999881 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 21.875, | |
| "completions/mean_terminated_length": 21.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07283683493733406, | |
| "epoch": 0.6328125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.019634408876299858, | |
| "kl": 0.007241100538522005, | |
| "learning_rate": 4.82777795418054e-06, | |
| "loss": 7.439294131472707e-05, | |
| "num_tokens": 388152.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 81, | |
| "step_time": 7.132434741999987 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 15.25, | |
| "completions/mean_terminated_length": 15.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08575013279914856, | |
| "epoch": 0.640625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03768326714634895, | |
| "kl": 0.010142359882593155, | |
| "learning_rate": 4.819378296439962e-06, | |
| "loss": 0.00010288630437571555, | |
| "num_tokens": 392866.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 82, | |
| "step_time": 6.954328571000133 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 13.0, | |
| "completions/max_terminated_length": 13.0, | |
| "completions/mean_length": 13.0, | |
| "completions/mean_terminated_length": 13.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12997420877218246, | |
| "epoch": 0.6484375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01888282783329487, | |
| "kl": 0.005712541053071618, | |
| "learning_rate": 4.810786316178377e-06, | |
| "loss": 5.712541315006092e-05, | |
| "num_tokens": 398394.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 83, | |
| "step_time": 6.931600935000006 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 22.125, | |
| "completions/mean_terminated_length": 22.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06609366461634636, | |
| "epoch": 0.65625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.017994992434978485, | |
| "kl": 0.00510314735583961, | |
| "learning_rate": 4.802002725841577e-06, | |
| "loss": 4.8029709432739764e-05, | |
| "num_tokens": 403295.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 84, | |
| "step_time": 7.442147588000125 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 42.0, | |
| "completions/max_terminated_length": 42.0, | |
| "completions/mean_length": 18.625, | |
| "completions/mean_terminated_length": 18.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11900537833571434, | |
| "epoch": 0.6640625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.024866754189133644, | |
| "kl": 0.006785314995795488, | |
| "learning_rate": 4.793028253763633e-06, | |
| "loss": 7.137990178307518e-05, | |
| "num_tokens": 408052.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 85, | |
| "step_time": 7.643961140000329 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 16.125, | |
| "completions/mean_terminated_length": 16.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.10917261242866516, | |
| "epoch": 0.671875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.022167593240737915, | |
| "kl": 0.0044796590227633715, | |
| "learning_rate": 4.783863644106502e-06, | |
| "loss": 4.586868089972995e-05, | |
| "num_tokens": 413581.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 86, | |
| "step_time": 7.9875519340000665 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06199129670858383, | |
| "epoch": 0.6796875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03173369914293289, | |
| "kl": 0.008890938945114613, | |
| "learning_rate": 4.774509656798326e-06, | |
| "loss": 8.7326108769048e-05, | |
| "num_tokens": 418435.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 87, | |
| "step_time": 6.503118833999906 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 25.375, | |
| "completions/mean_terminated_length": 25.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0360421109944582, | |
| "epoch": 0.6875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01023577805608511, | |
| "kl": 0.003528562723658979, | |
| "learning_rate": 4.764967067470409e-06, | |
| "loss": 3.279188240412623e-05, | |
| "num_tokens": 422510.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 88, | |
| "step_time": 5.7346701770002255 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.25, | |
| "completions/mean_terminated_length": 13.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13792025297880173, | |
| "epoch": 0.6953125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.036804474890232086, | |
| "kl": 0.007783198030665517, | |
| "learning_rate": 4.755236667392914e-06, | |
| "loss": 7.783197361277416e-05, | |
| "num_tokens": 428016.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 89, | |
| "step_time": 5.881843299000138 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 29.0, | |
| "completions/mean_terminated_length": 29.0, | |
| "completions/min_length": 29.0, | |
| "completions/min_terminated_length": 29.0, | |
| "entropy": 0.033070383593440056, | |
| "epoch": 0.703125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.002256783191114664, | |
| "kl": 0.001209279813338071, | |
| "learning_rate": 4.745319263409241e-06, | |
| "loss": 1.2092797987861559e-05, | |
| "num_tokens": 432116.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 90, | |
| "step_time": 5.656249395999794 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.125, | |
| "completions/max_length": 128.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 36.25, | |
| "completions/mean_terminated_length": 23.142858505249023, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.2254822440445423, | |
| "epoch": 0.7109375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.014093738049268723, | |
| "kl": 0.0030488879419863224, | |
| "learning_rate": 4.735215677869129e-06, | |
| "loss": 3.228533023502678e-05, | |
| "num_tokens": 437806.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 91, | |
| "step_time": 14.648245948000067 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 26.75, | |
| "completions/mean_terminated_length": 26.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05311024375259876, | |
| "epoch": 0.71875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00612290482968092, | |
| "kl": 0.0022808221401646733, | |
| "learning_rate": 4.724926748560464e-06, | |
| "loss": 2.2808220819570124e-05, | |
| "num_tokens": 442648.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 92, | |
| "step_time": 7.179930319999812 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 19.375, | |
| "completions/mean_terminated_length": 19.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11795984953641891, | |
| "epoch": 0.7265625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.014634083956480026, | |
| "kl": 0.0032170950435101986, | |
| "learning_rate": 4.714453328639814e-06, | |
| "loss": 3.587050741771236e-05, | |
| "num_tokens": 448179.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 93, | |
| "step_time": 7.483695861999877 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 19.875, | |
| "completions/mean_terminated_length": 19.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07148051261901855, | |
| "epoch": 0.734375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.010934879072010517, | |
| "kl": 0.0029092643526382744, | |
| "learning_rate": 4.7037962865616795e-06, | |
| "loss": 2.7483838493935764e-05, | |
| "num_tokens": 453030.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 94, | |
| "step_time": 7.076518674999988 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 22.125, | |
| "completions/mean_terminated_length": 22.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06901054829359055, | |
| "epoch": 0.7421875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006595511920750141, | |
| "kl": 0.0014058846863918006, | |
| "learning_rate": 4.692956506006486e-06, | |
| "loss": 1.403903206664836e-05, | |
| "num_tokens": 457855.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 95, | |
| "step_time": 7.476475853000011 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.125, | |
| "completions/mean_terminated_length": 17.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08090204373002052, | |
| "epoch": 0.75, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.020713720470666885, | |
| "kl": 0.005518489051610231, | |
| "learning_rate": 4.681934885807307e-06, | |
| "loss": 5.53158279217314e-05, | |
| "num_tokens": 462588.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 96, | |
| "step_time": 6.740178639000078 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 23.0, | |
| "completions/mean_terminated_length": 23.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.037110449746251106, | |
| "epoch": 0.7578125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006115948315709829, | |
| "kl": 0.0022233356721699238, | |
| "learning_rate": 4.6707323398753346e-06, | |
| "loss": 2.206304998253472e-05, | |
| "num_tokens": 466604.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 97, | |
| "step_time": 5.532072194999955 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06933313608169556, | |
| "epoch": 0.765625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007001116406172514, | |
| "kl": 0.0019759326823987067, | |
| "learning_rate": 4.659349797124096e-06, | |
| "loss": 1.9955001334892586e-05, | |
| "num_tokens": 471382.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 98, | |
| "step_time": 6.4825494700000945 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 25.5, | |
| "completions/mean_terminated_length": 25.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03348589688539505, | |
| "epoch": 0.7734375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.002566123381257057, | |
| "kl": 0.0018573449924588203, | |
| "learning_rate": 4.647788201392429e-06, | |
| "loss": 1.857344977906905e-05, | |
| "num_tokens": 475554.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 99, | |
| "step_time": 5.890160061000188 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.25, | |
| "completions/max_length": 128.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 42.0, | |
| "completions/mean_terminated_length": 13.333333969116211, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.21250910684466362, | |
| "epoch": 0.78125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.07006637006998062, | |
| "kl": 0.017085027415305376, | |
| "learning_rate": 4.636048511366222e-06, | |
| "loss": 0.00017085025319829583, | |
| "num_tokens": 481290.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 100, | |
| "step_time": 14.571886820999907 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 16.25, | |
| "completions/mean_terminated_length": 16.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11931024491786957, | |
| "epoch": 0.7890625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0073622651398181915, | |
| "kl": 0.0012832598586101085, | |
| "learning_rate": 4.624131700498913e-06, | |
| "loss": 1.3542096894525457e-05, | |
| "num_tokens": 486820.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 101, | |
| "step_time": 7.506882940999958 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.125, | |
| "completions/max_length": 128.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 33.75, | |
| "completions/mean_terminated_length": 20.285715103149414, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05769490636885166, | |
| "epoch": 0.796875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.002226189011707902, | |
| "kl": 0.001102528884075582, | |
| "learning_rate": 4.612038756930778e-06, | |
| "loss": 8.816463378025219e-06, | |
| "num_tokens": 491814.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 102, | |
| "step_time": 14.50194374800003 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.375, | |
| "completions/mean_terminated_length": 27.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.032190028578042984, | |
| "epoch": 0.8046875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0019165624398738146, | |
| "kl": 0.0013525813119485974, | |
| "learning_rate": 4.599770683406992e-06, | |
| "loss": 1.3399311683315318e-05, | |
| "num_tokens": 495785.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 103, | |
| "step_time": 5.6954472240001905 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 22.125, | |
| "completions/mean_terminated_length": 22.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09382087737321854, | |
| "epoch": 0.8125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0029365946538746357, | |
| "kl": 0.002741978387348354, | |
| "learning_rate": 4.587328497194478e-06, | |
| "loss": 2.726353341131471e-05, | |
| "num_tokens": 501338.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 104, | |
| "step_time": 7.473332564999964 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 26.75, | |
| "completions/mean_terminated_length": 26.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05792428180575371, | |
| "epoch": 0.8203125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0022111833095550537, | |
| "kl": 0.0035839949268847704, | |
| "learning_rate": 4.5747132299975634e-06, | |
| "loss": 4.1183855501003563e-05, | |
| "num_tokens": 506124.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 105, | |
| "step_time": 7.173797810999986 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.625, | |
| "completions/mean_terminated_length": 21.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06740124337375164, | |
| "epoch": 0.828125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0050131166353821754, | |
| "kl": 0.0017226869240403175, | |
| "learning_rate": 4.561925927872421e-06, | |
| "loss": 1.599166716914624e-05, | |
| "num_tokens": 510953.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 106, | |
| "step_time": 6.916509818999884 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.125, | |
| "completions/mean_terminated_length": 21.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07052117213606834, | |
| "epoch": 0.8359375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007340835873037577, | |
| "kl": 0.0017220252193510532, | |
| "learning_rate": 4.548967651140341e-06, | |
| "loss": 1.7247453797608614e-05, | |
| "num_tokens": 515806.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 107, | |
| "step_time": 6.615679707000027 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 24.125, | |
| "completions/mean_terminated_length": 24.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06415174063295126, | |
| "epoch": 0.84375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0005465157446451485, | |
| "kl": 0.001015377274597995, | |
| "learning_rate": 4.5358394742998e-06, | |
| "loss": 1.1552911928447429e-05, | |
| "num_tokens": 520699.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 108, | |
| "step_time": 7.493365885000003 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.25, | |
| "completions/mean_terminated_length": 21.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0641067698597908, | |
| "epoch": 0.8515625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.010530305095016956, | |
| "kl": 0.0026437516789883375, | |
| "learning_rate": 4.522542485937369e-06, | |
| "loss": 2.6550886104814708e-05, | |
| "num_tokens": 525437.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 109, | |
| "step_time": 6.527635427999712 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 25.75, | |
| "completions/mean_terminated_length": 25.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0577393751591444, | |
| "epoch": 0.859375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008794880472123623, | |
| "kl": 0.0028055842267349362, | |
| "learning_rate": 4.509077788637446e-06, | |
| "loss": 2.7364983907318674e-05, | |
| "num_tokens": 530267.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 110, | |
| "step_time": 7.3437313919998815 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 24.0, | |
| "completions/mean_terminated_length": 24.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0586474034935236, | |
| "epoch": 0.8671875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005623109173029661, | |
| "kl": 0.0036008708993904293, | |
| "learning_rate": 4.4954464988908306e-06, | |
| "loss": 3.5230626963311806e-05, | |
| "num_tokens": 535131.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 111, | |
| "step_time": 7.285259039999801 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 16.375, | |
| "completions/mean_terminated_length": 16.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12351097911596298, | |
| "epoch": 0.875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01802302524447441, | |
| "kl": 0.0029585817828774452, | |
| "learning_rate": 4.481649747002146e-06, | |
| "loss": 3.03143824567087e-05, | |
| "num_tokens": 540638.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 112, | |
| "step_time": 7.425657803999911 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.625, | |
| "completions/mean_terminated_length": 19.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06710582785308361, | |
| "epoch": 0.8828125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0028362085577100515, | |
| "kl": 0.0015989196253940463, | |
| "learning_rate": 4.467688676996111e-06, | |
| "loss": 1.6505855455761775e-05, | |
| "num_tokens": 545519.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 113, | |
| "step_time": 6.964208683000152 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.5, | |
| "completions/mean_terminated_length": 19.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07923074252903461, | |
| "epoch": 0.890625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008551168255507946, | |
| "kl": 0.002287313051056117, | |
| "learning_rate": 4.4535644465226795e-06, | |
| "loss": 1.9634611817309633e-05, | |
| "num_tokens": 550319.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 114, | |
| "step_time": 6.80012675099988 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 20.0, | |
| "completions/mean_terminated_length": 20.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07435985282063484, | |
| "epoch": 0.8984375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.014601176604628563, | |
| "kl": 0.005440297303721309, | |
| "learning_rate": 4.43927822676105e-06, | |
| "loss": 4.615134821506217e-05, | |
| "num_tokens": 555091.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 115, | |
| "step_time": 7.212101119999943 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08148583956062794, | |
| "epoch": 0.90625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004445843398571014, | |
| "kl": 0.001686684088781476, | |
| "learning_rate": 4.424831202322548e-06, | |
| "loss": 1.5020732462289743e-05, | |
| "num_tokens": 559873.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 116, | |
| "step_time": 6.573965233000081 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 30.875, | |
| "completions/mean_terminated_length": 30.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05395838990807533, | |
| "epoch": 0.9140625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.002361060120165348, | |
| "kl": 0.0016205881838686764, | |
| "learning_rate": 4.410224571152402e-06, | |
| "loss": 1.6156718629645184e-05, | |
| "num_tokens": 564700.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 117, | |
| "step_time": 7.449374204000151 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 25.5, | |
| "completions/mean_terminated_length": 25.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03191390447318554, | |
| "epoch": 0.921875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.04630071669816971, | |
| "kl": 0.015495602798182517, | |
| "learning_rate": 4.395459544430407e-06, | |
| "loss": 0.00015495602565351874, | |
| "num_tokens": 568832.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 118, | |
| "step_time": 5.798940031999791 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.375, | |
| "completions/mean_terminated_length": 19.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08851146325469017, | |
| "epoch": 0.9296875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03349859267473221, | |
| "kl": 0.0069463985855691135, | |
| "learning_rate": 4.380537346470495e-06, | |
| "loss": 7.56498338887468e-05, | |
| "num_tokens": 573559.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 119, | |
| "step_time": 6.745357660999844 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.041911693289875984, | |
| "epoch": 0.9375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.003103650640696287, | |
| "kl": 0.0021437766263261437, | |
| "learning_rate": 4.3654592146192146e-06, | |
| "loss": 2.1203726646490395e-05, | |
| "num_tokens": 577735.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 120, | |
| "step_time": 5.838892162999855 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 24.5, | |
| "completions/mean_terminated_length": 24.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0581835713237524, | |
| "epoch": 0.9453125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0017389818094670773, | |
| "kl": 0.00121387725812383, | |
| "learning_rate": 4.35022639915313e-06, | |
| "loss": 1.2284486729186028e-05, | |
| "num_tokens": 582627.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 121, | |
| "step_time": 7.239668368000139 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 29.0, | |
| "completions/mean_terminated_length": 29.0, | |
| "completions/min_length": 29.0, | |
| "completions/min_terminated_length": 29.0, | |
| "entropy": 0.027326339855790138, | |
| "epoch": 0.953125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00015551889373455197, | |
| "kl": 0.0014788485132157803, | |
| "learning_rate": 4.334840163175152e-06, | |
| "loss": 1.4788485714234412e-05, | |
| "num_tokens": 586899.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 122, | |
| "step_time": 5.908869453000079 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.125, | |
| "completions/mean_terminated_length": 19.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08600620739161968, | |
| "epoch": 0.9609375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004444346763193607, | |
| "kl": 0.0013841381878592074, | |
| "learning_rate": 4.319301782509794e-06, | |
| "loss": 1.4433127944357693e-05, | |
| "num_tokens": 591680.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 123, | |
| "step_time": 6.542298487999915 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 29.75, | |
| "completions/mean_terminated_length": 29.75, | |
| "completions/min_length": 14.0, | |
| "completions/min_terminated_length": 14.0, | |
| "entropy": 0.05058911070227623, | |
| "epoch": 0.96875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.002049674978479743, | |
| "kl": 0.0011390995350666344, | |
| "learning_rate": 4.30361254559739e-06, | |
| "loss": 1.1433874533395283e-05, | |
| "num_tokens": 596474.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 124, | |
| "step_time": 7.415240068999992 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 20.25, | |
| "completions/mean_terminated_length": 20.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.14710266143083572, | |
| "epoch": 0.9765625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02711130492389202, | |
| "kl": 0.0030616128933615983, | |
| "learning_rate": 4.287773753387249e-06, | |
| "loss": 3.7990492273820564e-05, | |
| "num_tokens": 602012.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 125, | |
| "step_time": 7.445253116999993 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 40.0, | |
| "completions/max_terminated_length": 40.0, | |
| "completions/mean_length": 25.625, | |
| "completions/mean_terminated_length": 25.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.060312340036034584, | |
| "epoch": 0.984375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004181354772299528, | |
| "kl": 0.0010917744366452098, | |
| "learning_rate": 4.271786719229787e-06, | |
| "loss": 1.0562279385339934e-05, | |
| "num_tokens": 606797.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 126, | |
| "step_time": 7.632959243999949 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.125, | |
| "completions/max_length": 128.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 27.75, | |
| "completions/mean_terminated_length": 13.428571701049805, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.10784962400794029, | |
| "epoch": 0.9921875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 1.5359536409378052, | |
| "kl": 0.055240524452528916, | |
| "learning_rate": 4.255652768767619e-06, | |
| "loss": 0.0008335658931173384, | |
| "num_tokens": 612419.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 127, | |
| "step_time": 14.956502908999937 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 57.0, | |
| "completions/max_terminated_length": 57.0, | |
| "completions/mean_length": 32.125, | |
| "completions/mean_terminated_length": 32.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09433136507868767, | |
| "epoch": 1.0, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.026125041767954826, | |
| "kl": 0.005883891310077161, | |
| "learning_rate": 4.23937323982564e-06, | |
| "loss": 6.240967923076823e-05, | |
| "num_tokens": 617328.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 128, | |
| "step_time": 8.913310376000027 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 25.0, | |
| "completions/mean_terminated_length": 25.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03300720080733299, | |
| "epoch": 1.0078125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0028305482119321823, | |
| "kl": 0.001693042810074985, | |
| "learning_rate": 4.222949482300094e-06, | |
| "loss": 1.6930427591432817e-05, | |
| "num_tokens": 621356.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 129, | |
| "step_time": 5.800293608999937 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07437415793538094, | |
| "epoch": 1.015625, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 3.94826078414917, | |
| "kl": 2.5161777958273888, | |
| "learning_rate": 4.206382858046636e-06, | |
| "loss": -0.11797107756137848, | |
| "num_tokens": 626170.0, | |
| "reward": 0.887499988079071, | |
| "reward_std": 0.3181980550289154, | |
| "rewards/reward_fn/mean": 0.887499988079071, | |
| "rewards/reward_fn/std": 0.3181980550289154, | |
| "step": 130, | |
| "step_time": 6.945177734999788 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 17.375, | |
| "completions/mean_terminated_length": 17.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07938191294670105, | |
| "epoch": 1.0234375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00952488835901022, | |
| "kl": 0.001886414596810937, | |
| "learning_rate": 4.189674740767411e-06, | |
| "loss": 1.6817120922496542e-05, | |
| "num_tokens": 630937.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 131, | |
| "step_time": 6.6918233229998805 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.25, | |
| "completions/mean_terminated_length": 17.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07850120961666107, | |
| "epoch": 1.03125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.013484103605151176, | |
| "kl": 0.0045427161967381835, | |
| "learning_rate": 4.172826515897146e-06, | |
| "loss": 4.5427161239786074e-05, | |
| "num_tokens": 635691.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 132, | |
| "step_time": 6.658873882999842 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08399690128862858, | |
| "epoch": 1.0390625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.025045400485396385, | |
| "kl": 0.029486034996807575, | |
| "learning_rate": 4.15583958048827e-06, | |
| "loss": 0.00029239041032269597, | |
| "num_tokens": 640561.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 133, | |
| "step_time": 6.894192197999928 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.125, | |
| "completions/mean_terminated_length": 13.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13748392462730408, | |
| "epoch": 1.046875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01748264580965042, | |
| "kl": 0.001855719368904829, | |
| "learning_rate": 4.138715343095069e-06, | |
| "loss": 1.858307223301381e-05, | |
| "num_tokens": 646038.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 134, | |
| "step_time": 5.438569751999921 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 23.125, | |
| "completions/mean_terminated_length": 23.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.04659535735845566, | |
| "epoch": 1.0546875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.051196254789829254, | |
| "kl": 0.01883502001874149, | |
| "learning_rate": 4.12145522365689e-06, | |
| "loss": 0.0001754522672854364, | |
| "num_tokens": 650127.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 135, | |
| "step_time": 5.736987513000258 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07493950799107552, | |
| "epoch": 1.0625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007082380820065737, | |
| "kl": 0.0019909495022147894, | |
| "learning_rate": 4.104060653380403e-06, | |
| "loss": 2.0618410417228006e-05, | |
| "num_tokens": 654983.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 136, | |
| "step_time": 6.561122189999878 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.125, | |
| "completions/mean_terminated_length": 17.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07711902260780334, | |
| "epoch": 1.0703125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.061376169323921204, | |
| "kl": 0.021036528050899506, | |
| "learning_rate": 4.086533074620919e-06, | |
| "loss": 0.0002116656833095476, | |
| "num_tokens": 659836.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 137, | |
| "step_time": 6.81616649800003 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 27.0, | |
| "completions/mean_terminated_length": 27.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.038892749696969986, | |
| "epoch": 1.078125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005187615752220154, | |
| "kl": 0.0026534507051110268, | |
| "learning_rate": 4.068873940762796e-06, | |
| "loss": 2.6154470106121153e-05, | |
| "num_tokens": 664044.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 138, | |
| "step_time": 5.881332467999982 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 15.625, | |
| "completions/mean_terminated_length": 15.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1011301763355732, | |
| "epoch": 1.0859375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.2945476174354553, | |
| "kl": 0.08709336444735527, | |
| "learning_rate": 4.051084716098921e-06, | |
| "loss": 0.0008627058705314994, | |
| "num_tokens": 668817.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 139, | |
| "step_time": 6.639636123999935 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.625, | |
| "completions/mean_terminated_length": 13.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12177715450525284, | |
| "epoch": 1.09375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.033345598727464676, | |
| "kl": 0.005186434369534254, | |
| "learning_rate": 4.033166875709291e-06, | |
| "loss": 5.185226473258808e-05, | |
| "num_tokens": 674298.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 140, | |
| "step_time": 5.517028535000009 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.5, | |
| "completions/mean_terminated_length": 17.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07133128866553307, | |
| "epoch": 1.1015625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.12929074466228485, | |
| "kl": 0.09029890596866608, | |
| "learning_rate": 4.015121905338704e-06, | |
| "loss": 0.0009029890061356127, | |
| "num_tokens": 678246.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 141, | |
| "step_time": 5.957433755000011 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.057012878358364105, | |
| "epoch": 1.109375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0789097398519516, | |
| "kl": 0.03356556943617761, | |
| "learning_rate": 3.996951301273556e-06, | |
| "loss": 0.0003683689865283668, | |
| "num_tokens": 682300.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 142, | |
| "step_time": 5.954758885000047 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.125, | |
| "completions/mean_terminated_length": 19.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07221270352602005, | |
| "epoch": 1.1171875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.08531598001718521, | |
| "kl": 0.04459764843340963, | |
| "learning_rate": 3.9786565702177725e-06, | |
| "loss": 0.00040393066592514515, | |
| "num_tokens": 687145.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 143, | |
| "step_time": 6.8996876880000855 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.25, | |
| "completions/mean_terminated_length": 13.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.14312979578971863, | |
| "epoch": 1.125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.020980026572942734, | |
| "kl": 0.03795890283072367, | |
| "learning_rate": 3.960239229167869e-06, | |
| "loss": 0.0003864379250444472, | |
| "num_tokens": 692651.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 144, | |
| "step_time": 5.689572013000088 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.125, | |
| "completions/mean_terminated_length": 13.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13573984801769257, | |
| "epoch": 1.1328125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.014293034560978413, | |
| "kl": 0.003559689619578421, | |
| "learning_rate": 3.941700805287169e-06, | |
| "loss": 3.578899850253947e-05, | |
| "num_tokens": 698156.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 145, | |
| "step_time": 5.635921549999921 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.75, | |
| "completions/mean_terminated_length": 19.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07221874222159386, | |
| "epoch": 1.140625, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 2.1729769706726074, | |
| "kl": 0.006373312848154455, | |
| "learning_rate": 3.92304283577916e-06, | |
| "loss": -0.15180744230747223, | |
| "num_tokens": 703038.0, | |
| "reward": 0.8812500238418579, | |
| "reward_std": 0.3358757197856903, | |
| "rewards/reward_fn/mean": 0.8812500238418579, | |
| "rewards/reward_fn/std": 0.3358757197856903, | |
| "step": 146, | |
| "step_time": 6.94864533499981 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.625, | |
| "completions/mean_terminated_length": 19.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06784151494503021, | |
| "epoch": 1.1484375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008188405074179173, | |
| "kl": 0.005511927942279726, | |
| "learning_rate": 3.904266867760044e-06, | |
| "loss": 5.116416286909953e-05, | |
| "num_tokens": 707899.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 147, | |
| "step_time": 6.646591747999992 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.5, | |
| "completions/mean_terminated_length": 19.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07835190743207932, | |
| "epoch": 1.15625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004332108423113823, | |
| "kl": 0.0011774248559959233, | |
| "learning_rate": 3.8853744581304376e-06, | |
| "loss": 1.2425527529558167e-05, | |
| "num_tokens": 712639.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 148, | |
| "step_time": 6.73792784200009 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.375, | |
| "completions/mean_terminated_length": 19.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07140998169779778, | |
| "epoch": 1.1640625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006420095916837454, | |
| "kl": 0.008202763739973307, | |
| "learning_rate": 3.866367173446281e-06, | |
| "loss": 7.896333409007639e-05, | |
| "num_tokens": 717406.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 149, | |
| "step_time": 6.581852653999931 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.125, | |
| "completions/mean_terminated_length": 17.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07744136825203896, | |
| "epoch": 1.171875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006486960221081972, | |
| "kl": 0.0021612545242533088, | |
| "learning_rate": 3.84724658978894e-06, | |
| "loss": 2.1575528080575168e-05, | |
| "num_tokens": 722111.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 150, | |
| "step_time": 6.601302765000128 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.5, | |
| "completions/mean_terminated_length": 13.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12781532481312752, | |
| "epoch": 1.1796875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.011086604557931423, | |
| "kl": 0.0017027303110808134, | |
| "learning_rate": 3.828014292634508e-06, | |
| "loss": 1.7027303329086863e-05, | |
| "num_tokens": 727595.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 151, | |
| "step_time": 5.656125526000324 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.75, | |
| "completions/mean_terminated_length": 19.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07872535288333893, | |
| "epoch": 1.1875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.030335335060954094, | |
| "kl": 0.017175353597849607, | |
| "learning_rate": 3.808671876722357e-06, | |
| "loss": 0.00016482984938193113, | |
| "num_tokens": 732313.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 152, | |
| "step_time": 6.728792943000144 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 23.25, | |
| "completions/mean_terminated_length": 23.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06778555735945702, | |
| "epoch": 1.1953125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005629899445921183, | |
| "kl": 0.001822992053348571, | |
| "learning_rate": 3.7892209459228802e-06, | |
| "loss": 1.8229919078294188e-05, | |
| "num_tokens": 737211.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 153, | |
| "step_time": 7.577010963999783 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 25.375, | |
| "completions/mean_terminated_length": 25.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03097234107553959, | |
| "epoch": 1.203125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00039684277726337314, | |
| "kl": 0.0014657879946753383, | |
| "learning_rate": 3.769663113104516e-06, | |
| "loss": 1.4650184311904013e-05, | |
| "num_tokens": 741306.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 154, | |
| "step_time": 5.836096522999696 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.75, | |
| "completions/mean_terminated_length": 13.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1351180225610733, | |
| "epoch": 1.2109375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.04005980119109154, | |
| "kl": 0.012403905857354403, | |
| "learning_rate": 3.7500000000000005e-06, | |
| "loss": 0.00012463353050407022, | |
| "num_tokens": 746788.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 155, | |
| "step_time": 5.403199547000213 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 25.25, | |
| "completions/mean_terminated_length": 25.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05965032987296581, | |
| "epoch": 1.21875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007664468139410019, | |
| "kl": 0.0025948547408916056, | |
| "learning_rate": 3.7302332370718988e-06, | |
| "loss": 2.4826764274621382e-05, | |
| "num_tokens": 751706.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 156, | |
| "step_time": 7.5867298230000415 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 23.375, | |
| "completions/mean_terminated_length": 23.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03286047466099262, | |
| "epoch": 1.2265625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0014668001094833016, | |
| "kl": 0.0018113566911779344, | |
| "learning_rate": 3.7103644633774015e-06, | |
| "loss": 1.7852235032478347e-05, | |
| "num_tokens": 755809.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 157, | |
| "step_time": 5.844333027999937 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.0, | |
| "completions/mean_terminated_length": 17.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07944399863481522, | |
| "epoch": 1.234375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00917236041277647, | |
| "kl": 0.0026575025403872132, | |
| "learning_rate": 3.690395326432421e-06, | |
| "loss": 2.6575022275210358e-05, | |
| "num_tokens": 760533.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 158, | |
| "step_time": 6.835314456999868 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.375, | |
| "completions/mean_terminated_length": 19.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07627736404538155, | |
| "epoch": 1.2421875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.016258908435702324, | |
| "kl": 0.003968668053857982, | |
| "learning_rate": 3.6703274820749736e-06, | |
| "loss": 3.887472485075705e-05, | |
| "num_tokens": 765320.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 159, | |
| "step_time": 6.874671181000167 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.375, | |
| "completions/mean_terminated_length": 19.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06996462866663933, | |
| "epoch": 1.25, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.011779413558542728, | |
| "kl": 0.002461543306708336, | |
| "learning_rate": 3.650162594327881e-06, | |
| "loss": 2.4944250981207006e-05, | |
| "num_tokens": 770199.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 160, | |
| "step_time": 6.942916448000005 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 29.5, | |
| "completions/mean_terminated_length": 29.5, | |
| "completions/min_length": 29.0, | |
| "completions/min_terminated_length": 29.0, | |
| "entropy": 0.02681659162044525, | |
| "epoch": 1.2578125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00040976243326440454, | |
| "kl": 0.0014975924859754741, | |
| "learning_rate": 3.6299023352607894e-06, | |
| "loss": 1.4975924386817496e-05, | |
| "num_tokens": 774395.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 161, | |
| "step_time": 6.015989045999959 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 26.875, | |
| "completions/mean_terminated_length": 26.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.061081623658537865, | |
| "epoch": 1.265625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.014576679095625877, | |
| "kl": 0.0038794887950643897, | |
| "learning_rate": 3.6095483848515223e-06, | |
| "loss": 3.555541479727253e-05, | |
| "num_tokens": 779170.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 162, | |
| "step_time": 7.245200251999904 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 22.125, | |
| "completions/mean_terminated_length": 22.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06859908811748028, | |
| "epoch": 1.2734375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.020452730357646942, | |
| "kl": 0.005623190430924296, | |
| "learning_rate": 3.589102430846773e-06, | |
| "loss": 5.230373062659055e-05, | |
| "num_tokens": 784051.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 163, | |
| "step_time": 7.287538059000099 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 25.0, | |
| "completions/mean_terminated_length": 25.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.028452389873564243, | |
| "epoch": 1.28125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004462169948965311, | |
| "kl": 0.001680202956777066, | |
| "learning_rate": 3.5685661686221644e-06, | |
| "loss": 1.6027215679059736e-05, | |
| "num_tokens": 788131.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 164, | |
| "step_time": 5.703693580999925 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.5, | |
| "completions/mean_terminated_length": 19.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05600108206272125, | |
| "epoch": 1.2890625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.021526971831917763, | |
| "kl": 0.006003132904879749, | |
| "learning_rate": 3.5479413010416606e-06, | |
| "loss": 5.7515826483722776e-05, | |
| "num_tokens": 792967.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 165, | |
| "step_time": 6.575796513000114 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.5, | |
| "completions/mean_terminated_length": 27.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.02504791971296072, | |
| "epoch": 1.296875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0030265813693404198, | |
| "kl": 0.0021526971249841154, | |
| "learning_rate": 3.527229538316371e-06, | |
| "loss": 2.115210190822836e-05, | |
| "num_tokens": 797183.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 166, | |
| "step_time": 5.972162173999777 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.375, | |
| "completions/mean_terminated_length": 13.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13807643204927444, | |
| "epoch": 1.3046875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03702490031719208, | |
| "kl": 0.006319678155705333, | |
| "learning_rate": 3.5064325978627365e-06, | |
| "loss": 6.32827723165974e-05, | |
| "num_tokens": 802710.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 167, | |
| "step_time": 5.624996752000243 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.875, | |
| "completions/mean_terminated_length": 27.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.02927943505346775, | |
| "epoch": 1.3125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0027886745519936085, | |
| "kl": 0.001402048219460994, | |
| "learning_rate": 3.4855522041601265e-06, | |
| "loss": 1.3811075405101292e-05, | |
| "num_tokens": 806769.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 168, | |
| "step_time": 5.729365316999974 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 22.875, | |
| "completions/mean_terminated_length": 22.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09153914824128151, | |
| "epoch": 1.3203125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.031587954610586166, | |
| "kl": 0.006058453815057874, | |
| "learning_rate": 3.4645900886078388e-06, | |
| "loss": 6.320517422864214e-05, | |
| "num_tokens": 812348.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 169, | |
| "step_time": 7.63299943700008 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.5, | |
| "completions/mean_terminated_length": 21.5, | |
| "completions/min_length": 14.0, | |
| "completions/min_terminated_length": 14.0, | |
| "entropy": 0.05722755752503872, | |
| "epoch": 1.328125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.014601275324821472, | |
| "kl": 0.003756172431167215, | |
| "learning_rate": 3.443547989381536e-06, | |
| "loss": 3.436290717218071e-05, | |
| "num_tokens": 817204.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 170, | |
| "step_time": 6.65200465199996 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09964188188314438, | |
| "epoch": 1.3359375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.06053476780653, | |
| "kl": 0.006807157536968589, | |
| "learning_rate": 3.422427651289118e-06, | |
| "loss": 6.788011523894966e-05, | |
| "num_tokens": 822734.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 171, | |
| "step_time": 7.496788298999945 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 29.625, | |
| "completions/mean_terminated_length": 29.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05451471544802189, | |
| "epoch": 1.34375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008337209932506084, | |
| "kl": 0.0015613330178894103, | |
| "learning_rate": 3.4012308256260366e-06, | |
| "loss": 1.5506457202718593e-05, | |
| "num_tokens": 827671.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 172, | |
| "step_time": 7.244489809999777 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.125, | |
| "completions/max_length": 128.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 38.875, | |
| "completions/mean_terminated_length": 26.142858505249023, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.6550341844558716, | |
| "epoch": 1.3515625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008499568328261375, | |
| "kl": 0.0018006233731284738, | |
| "learning_rate": 3.3799592700300867e-06, | |
| "loss": 1.8337512301513925e-05, | |
| "num_tokens": 833354.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 173, | |
| "step_time": 14.681682713999862 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 23.875, | |
| "completions/mean_terminated_length": 23.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.060140494257211685, | |
| "epoch": 1.359375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00928256381303072, | |
| "kl": 0.0029123042477294803, | |
| "learning_rate": 3.3586147483356534e-06, | |
| "loss": 2.732695429585874e-05, | |
| "num_tokens": 838109.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 174, | |
| "step_time": 7.2412520599998516 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 40.0, | |
| "completions/max_terminated_length": 40.0, | |
| "completions/mean_length": 24.375, | |
| "completions/mean_terminated_length": 24.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06309142522513866, | |
| "epoch": 1.3671875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00517475139349699, | |
| "kl": 0.0014793115551583469, | |
| "learning_rate": 3.3371990304274654e-06, | |
| "loss": 1.494552179792663e-05, | |
| "num_tokens": 843028.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 175, | |
| "step_time": 7.81271701799983 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.0, | |
| "completions/mean_terminated_length": 21.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03291504085063934, | |
| "epoch": 1.375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004775932524353266, | |
| "kl": 0.002072959439828992, | |
| "learning_rate": 3.315713892093829e-06, | |
| "loss": 2.0729594325530343e-05, | |
| "num_tokens": 847064.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 176, | |
| "step_time": 5.776862065000159 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 35.0, | |
| "completions/max_terminated_length": 35.0, | |
| "completions/mean_length": 16.25, | |
| "completions/mean_terminated_length": 16.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13138685747981071, | |
| "epoch": 1.3828125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.023814553394913673, | |
| "kl": 0.0031949164113029838, | |
| "learning_rate": 3.294161114879382e-06, | |
| "loss": 3.185410605510697e-05, | |
| "num_tokens": 852594.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 177, | |
| "step_time": 7.431254295000144 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 16.25, | |
| "completions/mean_terminated_length": 16.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12037009745836258, | |
| "epoch": 1.390625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.018406132236123085, | |
| "kl": 0.003100107191130519, | |
| "learning_rate": 3.272542485937369e-06, | |
| "loss": 3.313878914923407e-05, | |
| "num_tokens": 858124.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 178, | |
| "step_time": 7.8139032770000085 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 39.0, | |
| "completions/max_terminated_length": 39.0, | |
| "completions/mean_length": 28.625, | |
| "completions/mean_terminated_length": 28.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.16150045860558748, | |
| "epoch": 1.3984375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.24875901639461517, | |
| "kl": 0.045317163807339966, | |
| "learning_rate": 3.2508597978814515e-06, | |
| "loss": 0.00043976842425763607, | |
| "num_tokens": 862249.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 179, | |
| "step_time": 6.564737340999727 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.5, | |
| "completions/mean_terminated_length": 27.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03088196087628603, | |
| "epoch": 1.40625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0007541459635831416, | |
| "kl": 0.0011516560916788876, | |
| "learning_rate": 3.2291148486370626e-06, | |
| "loss": 1.1446018106653355e-05, | |
| "num_tokens": 866265.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 180, | |
| "step_time": 5.712320224999758 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.75, | |
| "completions/mean_terminated_length": 21.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06882932037115097, | |
| "epoch": 1.4140625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00206294865347445, | |
| "kl": 0.0013715573586523533, | |
| "learning_rate": 3.207309441292325e-06, | |
| "loss": 1.3715573004446924e-05, | |
| "num_tokens": 871011.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 181, | |
| "step_time": 6.726993576000041 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 90.0, | |
| "completions/max_terminated_length": 90.0, | |
| "completions/mean_length": 29.125, | |
| "completions/mean_terminated_length": 29.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.5393692404031754, | |
| "epoch": 1.421875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02723606675863266, | |
| "kl": 0.0031313110375776887, | |
| "learning_rate": 3.185445383948539e-06, | |
| "loss": 3.407296753721312e-05, | |
| "num_tokens": 876640.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 182, | |
| "step_time": 11.819733140000153 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 20.75, | |
| "completions/mean_terminated_length": 20.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12167311646044254, | |
| "epoch": 1.4296875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02585308626294136, | |
| "kl": 0.002457446011248976, | |
| "learning_rate": 3.1635244895702527e-06, | |
| "loss": 2.4284261598950252e-05, | |
| "num_tokens": 881406.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 183, | |
| "step_time": 6.903921601999855 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 23.0, | |
| "completions/mean_terminated_length": 23.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06957883015275002, | |
| "epoch": 1.4375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0076453592628240585, | |
| "kl": 0.001370629877783358, | |
| "learning_rate": 3.1415485758349344e-06, | |
| "loss": 1.3765486073680222e-05, | |
| "num_tokens": 886226.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 184, | |
| "step_time": 7.334037624999837 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.5, | |
| "completions/mean_terminated_length": 19.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0735643245279789, | |
| "epoch": 1.4453125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005766255781054497, | |
| "kl": 0.0016721467836759984, | |
| "learning_rate": 3.11951946498225e-06, | |
| "loss": 1.5945741324685514e-05, | |
| "num_tokens": 890954.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 185, | |
| "step_time": 6.7520039959997575 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 77.0, | |
| "completions/max_terminated_length": 77.0, | |
| "completions/mean_length": 33.375, | |
| "completions/mean_terminated_length": 33.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.043096503242850304, | |
| "epoch": 1.453125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004805476404726505, | |
| "kl": 0.001356584718450904, | |
| "learning_rate": 3.0974389836629628e-06, | |
| "loss": 1.3770947589364368e-05, | |
| "num_tokens": 895109.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 186, | |
| "step_time": 9.572878103999756 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 16.375, | |
| "completions/mean_terminated_length": 16.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1160760298371315, | |
| "epoch": 1.4609375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.021316073834896088, | |
| "kl": 0.0027994479751214385, | |
| "learning_rate": 3.0753089627874668e-06, | |
| "loss": 2.722722274484113e-05, | |
| "num_tokens": 900640.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 187, | |
| "step_time": 7.767691414999717 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 25.0, | |
| "completions/mean_terminated_length": 25.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.034027645364403725, | |
| "epoch": 1.46875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0012773608323186636, | |
| "kl": 0.0015459859278053045, | |
| "learning_rate": 3.0531312373739695e-06, | |
| "loss": 1.545986015116796e-05, | |
| "num_tokens": 904644.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 188, | |
| "step_time": 5.866847418999896 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 91.0, | |
| "completions/max_terminated_length": 91.0, | |
| "completions/mean_length": 31.75, | |
| "completions/mean_terminated_length": 31.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05926687642931938, | |
| "epoch": 1.4765625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02531527541577816, | |
| "kl": 0.0033996477723121643, | |
| "learning_rate": 3.030907646396333e-06, | |
| "loss": 3.2442520023323596e-05, | |
| "num_tokens": 909574.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 189, | |
| "step_time": 11.722558253999978 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 32.0, | |
| "completions/max_terminated_length": 32.0, | |
| "completions/mean_length": 21.875, | |
| "completions/mean_terminated_length": 21.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06604529172182083, | |
| "epoch": 1.484375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.014295347966253757, | |
| "kl": 0.004741971264593303, | |
| "learning_rate": 3.0086400326315853e-06, | |
| "loss": 4.750135849462822e-05, | |
| "num_tokens": 914337.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 190, | |
| "step_time": 6.908794395000086 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.0, | |
| "completions/mean_terminated_length": 21.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06935635954141617, | |
| "epoch": 1.4921875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0014209807850420475, | |
| "kl": 0.0011486579023767263, | |
| "learning_rate": 2.9863302425071156e-06, | |
| "loss": 1.2237365808687173e-05, | |
| "num_tokens": 919185.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 191, | |
| "step_time": 6.616372841999919 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 18.75, | |
| "completions/mean_terminated_length": 18.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11195787042379379, | |
| "epoch": 1.5, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.11325943470001221, | |
| "kl": 0.004804897769645322, | |
| "learning_rate": 2.963980125947573e-06, | |
| "loss": 6.231457518879324e-05, | |
| "num_tokens": 924735.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 192, | |
| "step_time": 7.530646898999976 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 35.0, | |
| "completions/max_terminated_length": 35.0, | |
| "completions/mean_length": 15.875, | |
| "completions/mean_terminated_length": 15.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1282949000597, | |
| "epoch": 1.5078125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02611485868692398, | |
| "kl": 0.001775389551767148, | |
| "learning_rate": 2.941591536221469e-06, | |
| "loss": 2.0562109057209454e-05, | |
| "num_tokens": 930262.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 193, | |
| "step_time": 7.432371278000119 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 27.0, | |
| "completions/mean_terminated_length": 27.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.029940275475382805, | |
| "epoch": 1.515625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0013159937225282192, | |
| "kl": 0.001477666082791984, | |
| "learning_rate": 2.9191663297875027e-06, | |
| "loss": 1.4661342902400065e-05, | |
| "num_tokens": 934362.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 194, | |
| "step_time": 5.929965283999991 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.5, | |
| "completions/mean_terminated_length": 13.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.142668217420578, | |
| "epoch": 1.5234375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.05234305188059807, | |
| "kl": 0.005082390503957868, | |
| "learning_rate": 2.896706366140629e-06, | |
| "loss": 5.082390271127224e-05, | |
| "num_tokens": 939870.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 195, | |
| "step_time": 5.839425092999818 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.125, | |
| "completions/mean_terminated_length": 21.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06505489535629749, | |
| "epoch": 1.53125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.002731472020968795, | |
| "kl": 0.001581202377565205, | |
| "learning_rate": 2.8742135076578608e-06, | |
| "loss": 1.5823465219000354e-05, | |
| "num_tokens": 944735.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 196, | |
| "step_time": 6.905690054999923 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 19.375, | |
| "completions/mean_terminated_length": 19.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11012165620923042, | |
| "epoch": 1.5390625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0070010521449148655, | |
| "kl": 0.0010562781826592982, | |
| "learning_rate": 2.8516896194438515e-06, | |
| "loss": 1.0670170013327152e-05, | |
| "num_tokens": 950290.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 197, | |
| "step_time": 7.652169488000027 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 35.0, | |
| "completions/max_terminated_length": 35.0, | |
| "completions/mean_length": 21.875, | |
| "completions/mean_terminated_length": 21.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06521525606513023, | |
| "epoch": 1.546875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.017679838463664055, | |
| "kl": 0.0035706963972188532, | |
| "learning_rate": 2.8291365691762313e-06, | |
| "loss": 3.664140967885032e-05, | |
| "num_tokens": 955021.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 198, | |
| "step_time": 7.155417588999853 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 25.25, | |
| "completions/mean_terminated_length": 25.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.049559252336621284, | |
| "epoch": 1.5546875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.05125805735588074, | |
| "kl": 0.01641816832125187, | |
| "learning_rate": 2.8065562269507464e-06, | |
| "loss": 0.00016517679614480585, | |
| "num_tokens": 959027.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 199, | |
| "step_time": 5.624229268999898 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 26.875, | |
| "completions/mean_terminated_length": 26.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.04596647992730141, | |
| "epoch": 1.5625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.009446537122130394, | |
| "kl": 0.0029358353931456804, | |
| "learning_rate": 2.7839504651261873e-06, | |
| "loss": 3.024380566785112e-05, | |
| "num_tokens": 963106.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 200, | |
| "step_time": 5.7302313129998765 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 24.125, | |
| "completions/mean_terminated_length": 24.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06624070927500725, | |
| "epoch": 1.5703125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0009546372457407415, | |
| "kl": 0.0009917069983202964, | |
| "learning_rate": 2.761321158169134e-06, | |
| "loss": 1.0749818102340214e-05, | |
| "num_tokens": 968011.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 201, | |
| "step_time": 7.5783325940001305 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 23.125, | |
| "completions/mean_terminated_length": 23.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03553210385143757, | |
| "epoch": 1.578125, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.7552725076675415, | |
| "kl": 0.06742474652128294, | |
| "learning_rate": 2.7386701824985257e-06, | |
| "loss": -0.07762715220451355, | |
| "num_tokens": 972088.0, | |
| "reward": 0.8812500238418579, | |
| "reward_std": 0.3358757197856903, | |
| "rewards/reward_fn/mean": 0.8812500238418579, | |
| "rewards/reward_fn/std": 0.3358757197856903, | |
| "step": 202, | |
| "step_time": 5.655839926999988 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.125, | |
| "completions/mean_terminated_length": 21.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06611593812704086, | |
| "epoch": 1.5859375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.002574133686721325, | |
| "kl": 0.0014704846544191241, | |
| "learning_rate": 2.715999416330068e-06, | |
| "loss": 1.4717452359036542e-05, | |
| "num_tokens": 976941.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 203, | |
| "step_time": 6.626762887999803 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 16.375, | |
| "completions/mean_terminated_length": 16.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12001239135861397, | |
| "epoch": 1.59375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.022112533450126648, | |
| "kl": 0.004337176156695932, | |
| "learning_rate": 2.6933107395204926e-06, | |
| "loss": 3.791246490436606e-05, | |
| "num_tokens": 982472.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 204, | |
| "step_time": 7.7095311099999435 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.125, | |
| "completions/mean_terminated_length": 21.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.037428947165608406, | |
| "epoch": 1.6015625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005030359141528606, | |
| "kl": 0.002402382204309106, | |
| "learning_rate": 2.670606033411678e-06, | |
| "loss": 2.403911275905557e-05, | |
| "num_tokens": 986525.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 205, | |
| "step_time": 5.898442960000239 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 23.875, | |
| "completions/mean_terminated_length": 23.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06124606728553772, | |
| "epoch": 1.609375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.017411913722753525, | |
| "kl": 0.004249897378031164, | |
| "learning_rate": 2.6478871806746496e-06, | |
| "loss": 3.866076440317556e-05, | |
| "num_tokens": 991340.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 206, | |
| "step_time": 7.1736011519999465 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 24.0, | |
| "completions/mean_terminated_length": 24.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07086298987269402, | |
| "epoch": 1.6171875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007288594264537096, | |
| "kl": 0.0015791382757015526, | |
| "learning_rate": 2.625156065153473e-06, | |
| "loss": 1.550133674754761e-05, | |
| "num_tokens": 996212.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 207, | |
| "step_time": 7.194994807000057 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 18.875, | |
| "completions/mean_terminated_length": 18.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.10177021846175194, | |
| "epoch": 1.625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.012765838764607906, | |
| "kl": 0.003196087433025241, | |
| "learning_rate": 2.602414571709036e-06, | |
| "loss": 3.193688462488353e-05, | |
| "num_tokens": 1001759.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 208, | |
| "step_time": 7.412187181000036 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.25, | |
| "completions/mean_terminated_length": 13.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13615204393863678, | |
| "epoch": 1.6328125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01755693182349205, | |
| "kl": 0.0030227339593693614, | |
| "learning_rate": 2.5796645860627665e-06, | |
| "loss": 3.0503688321914524e-05, | |
| "num_tokens": 1007265.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 209, | |
| "step_time": 5.581415800000059 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 54.0, | |
| "completions/max_terminated_length": 54.0, | |
| "completions/mean_length": 28.75, | |
| "completions/mean_terminated_length": 28.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.15063174068927765, | |
| "epoch": 1.640625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007047915831208229, | |
| "kl": 0.002616984653286636, | |
| "learning_rate": 2.556907994640264e-06, | |
| "loss": 2.5744397134985775e-05, | |
| "num_tokens": 1011331.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 210, | |
| "step_time": 7.517502090000107 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.375, | |
| "completions/mean_terminated_length": 21.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05742892995476723, | |
| "epoch": 1.6484375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006198030896484852, | |
| "kl": 0.00281632284168154, | |
| "learning_rate": 2.5341466844148775e-06, | |
| "loss": 2.817007407429628e-05, | |
| "num_tokens": 1016182.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 211, | |
| "step_time": 6.626822569000069 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 67.0, | |
| "completions/max_terminated_length": 67.0, | |
| "completions/mean_length": 30.875, | |
| "completions/mean_terminated_length": 30.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06470214761793613, | |
| "epoch": 1.65625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007548889145255089, | |
| "kl": 0.0024077071575447917, | |
| "learning_rate": 2.511382542751239e-06, | |
| "loss": 2.439593299641274e-05, | |
| "num_tokens": 1021133.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 212, | |
| "step_time": 9.792585082999722 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.125, | |
| "completions/max_length": 128.0, | |
| "completions/max_terminated_length": 107.0, | |
| "completions/mean_length": 45.125, | |
| "completions/mean_terminated_length": 33.28571701049805, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08313845098018646, | |
| "epoch": 1.6640625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007846017368137836, | |
| "kl": 0.0020210095099173486, | |
| "learning_rate": 2.488617457248761e-06, | |
| "loss": 1.9834918930428103e-05, | |
| "num_tokens": 1026074.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 213, | |
| "step_time": 14.740446427000279 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 25.0, | |
| "completions/mean_terminated_length": 25.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03169822786003351, | |
| "epoch": 1.671875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004874436184763908, | |
| "kl": 0.0018856602837331593, | |
| "learning_rate": 2.465853315585123e-06, | |
| "loss": 1.8856600945582613e-05, | |
| "num_tokens": 1030230.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 214, | |
| "step_time": 5.843885092000164 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 41.0, | |
| "completions/max_terminated_length": 41.0, | |
| "completions/mean_length": 22.75, | |
| "completions/mean_terminated_length": 22.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11479110643267632, | |
| "epoch": 1.6796875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01489281002432108, | |
| "kl": 0.003329504979774356, | |
| "learning_rate": 2.443092005359736e-06, | |
| "loss": 3.363801079103723e-05, | |
| "num_tokens": 1034984.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 215, | |
| "step_time": 7.630429413000456 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 23.25, | |
| "completions/mean_terminated_length": 23.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06059539131820202, | |
| "epoch": 1.6875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.011219343170523643, | |
| "kl": 0.002763925353065133, | |
| "learning_rate": 2.420335413937234e-06, | |
| "loss": 2.7146501452079974e-05, | |
| "num_tokens": 1039826.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 216, | |
| "step_time": 7.484693694999805 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 24.125, | |
| "completions/mean_terminated_length": 24.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06515759788453579, | |
| "epoch": 1.6953125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006574144121259451, | |
| "kl": 0.0019047525711357594, | |
| "learning_rate": 2.3975854282909645e-06, | |
| "loss": 1.847416569944471e-05, | |
| "num_tokens": 1044679.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 217, | |
| "step_time": 7.4298257559999 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 45.0, | |
| "completions/max_terminated_length": 45.0, | |
| "completions/mean_length": 27.0, | |
| "completions/mean_terminated_length": 27.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05187015235424042, | |
| "epoch": 1.703125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008155517280101776, | |
| "kl": 0.002550492761656642, | |
| "learning_rate": 2.374843934846528e-06, | |
| "loss": 2.506848977645859e-05, | |
| "num_tokens": 1048823.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 218, | |
| "step_time": 7.006674997000118 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 15.875, | |
| "completions/mean_terminated_length": 15.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12386251240968704, | |
| "epoch": 1.7109375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.019120769575238228, | |
| "kl": 0.003044891287572682, | |
| "learning_rate": 2.35211281932535e-06, | |
| "loss": 3.0685067031299695e-05, | |
| "num_tokens": 1054326.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 219, | |
| "step_time": 7.376969552999526 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.5, | |
| "completions/mean_terminated_length": 13.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12342415750026703, | |
| "epoch": 1.71875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.017717929556965828, | |
| "kl": 0.004510085214860737, | |
| "learning_rate": 2.3293939665883233e-06, | |
| "loss": 4.5417702494887635e-05, | |
| "num_tokens": 1059834.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 220, | |
| "step_time": 5.555540719999954 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 40.0, | |
| "completions/max_terminated_length": 40.0, | |
| "completions/mean_length": 24.875, | |
| "completions/mean_terminated_length": 24.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05634631961584091, | |
| "epoch": 1.7265625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008950191549956799, | |
| "kl": 0.0023297353181988, | |
| "learning_rate": 2.306689260479508e-06, | |
| "loss": 2.305710586369969e-05, | |
| "num_tokens": 1064757.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 221, | |
| "step_time": 7.691754480999407 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.125, | |
| "completions/mean_terminated_length": 21.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06376663595438004, | |
| "epoch": 1.734375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007804343942552805, | |
| "kl": 0.0021846327581442893, | |
| "learning_rate": 2.284000583669933e-06, | |
| "loss": 2.1814143110532314e-05, | |
| "num_tokens": 1069550.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 222, | |
| "step_time": 6.525545030000103 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 90.0, | |
| "completions/max_terminated_length": 90.0, | |
| "completions/mean_length": 29.0, | |
| "completions/mean_terminated_length": 29.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08469109237194061, | |
| "epoch": 1.7421875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008673850446939468, | |
| "kl": 0.0024135791463777423, | |
| "learning_rate": 2.261329817501475e-06, | |
| "loss": 2.4780489184195176e-05, | |
| "num_tokens": 1074362.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 223, | |
| "step_time": 11.37344323599973 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07181186601519585, | |
| "epoch": 1.75, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00881747156381607, | |
| "kl": 0.0025651361793279648, | |
| "learning_rate": 2.238678841830867e-06, | |
| "loss": 2.4443201255053282e-05, | |
| "num_tokens": 1079154.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 224, | |
| "step_time": 6.7924440969995885 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.0, | |
| "completions/mean_terminated_length": 21.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07576226070523262, | |
| "epoch": 1.7578125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.009993416257202625, | |
| "kl": 0.0031048988457769156, | |
| "learning_rate": 2.2160495348738127e-06, | |
| "loss": 3.1577099434798583e-05, | |
| "num_tokens": 1084034.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 225, | |
| "step_time": 6.582048697000118 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06813675723969936, | |
| "epoch": 1.765625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.013243633322417736, | |
| "kl": 0.0029321471229195595, | |
| "learning_rate": 2.1934437730492544e-06, | |
| "loss": 2.889215829782188e-05, | |
| "num_tokens": 1088838.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 226, | |
| "step_time": 6.774607225000182 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 96.0, | |
| "completions/max_terminated_length": 96.0, | |
| "completions/mean_length": 27.625, | |
| "completions/mean_terminated_length": 27.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0714375339448452, | |
| "epoch": 1.7734375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006967134308069944, | |
| "kl": 0.0022760473075322807, | |
| "learning_rate": 2.1708634308237687e-06, | |
| "loss": 2.1496147383004427e-05, | |
| "num_tokens": 1093735.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 227, | |
| "step_time": 11.807745952999994 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 31.0, | |
| "completions/max_terminated_length": 31.0, | |
| "completions/mean_length": 19.375, | |
| "completions/mean_terminated_length": 19.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11793160066008568, | |
| "epoch": 1.78125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.05189337208867073, | |
| "kl": 0.014766670181415975, | |
| "learning_rate": 2.1483103805561493e-06, | |
| "loss": 0.00013642504927702248, | |
| "num_tokens": 1098522.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 228, | |
| "step_time": 6.672160937999706 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06688312441110611, | |
| "epoch": 1.7890625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.011055883020162582, | |
| "kl": 0.0036998813739046454, | |
| "learning_rate": 2.1257864923421405e-06, | |
| "loss": 3.6064928281120956e-05, | |
| "num_tokens": 1103364.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 229, | |
| "step_time": 6.56204488100002 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 114.0, | |
| "completions/max_terminated_length": 114.0, | |
| "completions/mean_length": 33.875, | |
| "completions/mean_terminated_length": 33.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.051064563915133476, | |
| "epoch": 1.796875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008055821992456913, | |
| "kl": 0.0023585216840729117, | |
| "learning_rate": 2.1032936338593716e-06, | |
| "loss": 2.3143395083025098e-05, | |
| "num_tokens": 1107471.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 230, | |
| "step_time": 12.61657659399998 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 83.0, | |
| "completions/max_terminated_length": 83.0, | |
| "completions/mean_length": 25.875, | |
| "completions/mean_terminated_length": 25.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0740746557712555, | |
| "epoch": 1.8046875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007505510468035936, | |
| "kl": 0.0018645531381480396, | |
| "learning_rate": 2.080833670212498e-06, | |
| "loss": 1.8865794118028134e-05, | |
| "num_tokens": 1112306.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 231, | |
| "step_time": 11.113060962999953 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.125, | |
| "completions/mean_terminated_length": 21.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0632933359593153, | |
| "epoch": 1.8125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005921315401792526, | |
| "kl": 0.0018682752852328122, | |
| "learning_rate": 2.0584084637785316e-06, | |
| "loss": 1.870766755018849e-05, | |
| "num_tokens": 1117095.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 232, | |
| "step_time": 6.704811484000402 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 40.0, | |
| "completions/max_terminated_length": 40.0, | |
| "completions/mean_length": 23.875, | |
| "completions/mean_terminated_length": 23.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08165674656629562, | |
| "epoch": 1.8203125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.012375043705105782, | |
| "kl": 0.0033231035922653973, | |
| "learning_rate": 2.036019874052428e-06, | |
| "loss": 3.117723827017471e-05, | |
| "num_tokens": 1121946.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 233, | |
| "step_time": 7.801074556999993 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 31.0, | |
| "completions/max_terminated_length": 31.0, | |
| "completions/mean_length": 19.5, | |
| "completions/mean_terminated_length": 19.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08600634709000587, | |
| "epoch": 1.828125, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 1.3644095659255981, | |
| "kl": 0.06359560182318091, | |
| "learning_rate": 2.0136697574928853e-06, | |
| "loss": 0.06466268002986908, | |
| "num_tokens": 1126706.0, | |
| "reward": 0.875, | |
| "reward_std": 0.3535533845424652, | |
| "rewards/reward_fn/mean": 0.875, | |
| "rewards/reward_fn/std": 0.3535533845424652, | |
| "step": 234, | |
| "step_time": 6.896724009000081 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 43.0, | |
| "completions/max_terminated_length": 43.0, | |
| "completions/mean_length": 25.375, | |
| "completions/mean_terminated_length": 25.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08186899498105049, | |
| "epoch": 1.8359375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.011324622668325901, | |
| "kl": 0.003795566619373858, | |
| "learning_rate": 1.991359967368416e-06, | |
| "loss": 3.484051558189094e-05, | |
| "num_tokens": 1131529.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 235, | |
| "step_time": 7.659685747999902 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.125, | |
| "completions/mean_terminated_length": 17.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.072533143684268, | |
| "epoch": 1.84375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.015638258308172226, | |
| "kl": 0.003511433256790042, | |
| "learning_rate": 1.9690923536036673e-06, | |
| "loss": 3.5228476917836815e-05, | |
| "num_tokens": 1136314.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 236, | |
| "step_time": 6.76509731599981 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.0, | |
| "completions/mean_terminated_length": 21.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0642678551375866, | |
| "epoch": 1.8515625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006433664355427027, | |
| "kl": 0.0019541659858077765, | |
| "learning_rate": 1.9468687626260314e-06, | |
| "loss": 1.9541659639799036e-05, | |
| "num_tokens": 1141102.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 237, | |
| "step_time": 6.493211311999858 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 23.0, | |
| "completions/mean_terminated_length": 23.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03841168060898781, | |
| "epoch": 1.859375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.014958206564188004, | |
| "kl": 0.004870379460044205, | |
| "learning_rate": 1.9246910372125345e-06, | |
| "loss": 4.770414670929313e-05, | |
| "num_tokens": 1145026.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 238, | |
| "step_time": 5.479256311999961 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 40.0, | |
| "completions/max_terminated_length": 40.0, | |
| "completions/mean_length": 16.75, | |
| "completions/mean_terminated_length": 16.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11552144214510918, | |
| "epoch": 1.8671875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.06090380623936653, | |
| "kl": 0.012604215648025274, | |
| "learning_rate": 1.9025610163370385e-06, | |
| "loss": 0.0001225811429321766, | |
| "num_tokens": 1150584.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 239, | |
| "step_time": 7.893678880999687 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.0, | |
| "completions/mean_terminated_length": 21.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06876265630126, | |
| "epoch": 1.875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.009387739934027195, | |
| "kl": 0.003080976544879377, | |
| "learning_rate": 1.8804805350177507e-06, | |
| "loss": 2.9121343686711043e-05, | |
| "num_tokens": 1155364.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 240, | |
| "step_time": 6.4674589349997404 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.375, | |
| "completions/mean_terminated_length": 13.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.14056620001792908, | |
| "epoch": 1.8828125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.051601000130176544, | |
| "kl": 0.009546731133013964, | |
| "learning_rate": 1.8584514241650667e-06, | |
| "loss": 9.567832603352144e-05, | |
| "num_tokens": 1160847.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 241, | |
| "step_time": 5.5828178320002735 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.875, | |
| "completions/mean_terminated_length": 27.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.036137545481324196, | |
| "epoch": 1.890625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.009184165857732296, | |
| "kl": 0.0031361767323687673, | |
| "learning_rate": 1.8364755104297477e-06, | |
| "loss": 3.063581243623048e-05, | |
| "num_tokens": 1165038.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 242, | |
| "step_time": 5.927890447000209 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 40.0, | |
| "completions/max_terminated_length": 40.0, | |
| "completions/mean_length": 22.75, | |
| "completions/mean_terminated_length": 22.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07793265208601952, | |
| "epoch": 1.8984375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02107059955596924, | |
| "kl": 0.006186584243550897, | |
| "learning_rate": 1.8145546160514622e-06, | |
| "loss": 6.0147060139570385e-05, | |
| "num_tokens": 1169948.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 243, | |
| "step_time": 7.625320269999975 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 13.0, | |
| "completions/max_terminated_length": 13.0, | |
| "completions/mean_length": 13.0, | |
| "completions/mean_terminated_length": 13.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12505637109279633, | |
| "epoch": 1.90625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.020931562408804893, | |
| "kl": 0.004113408271223307, | |
| "learning_rate": 1.792690558707675e-06, | |
| "loss": 4.1134080674964935e-05, | |
| "num_tokens": 1175424.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 244, | |
| "step_time": 5.521570587000042 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.25, | |
| "completions/mean_terminated_length": 21.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.043781496584415436, | |
| "epoch": 1.9140625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.025867925956845284, | |
| "kl": 0.010342990979552269, | |
| "learning_rate": 1.7708851513629376e-06, | |
| "loss": 9.541432518744841e-05, | |
| "num_tokens": 1179346.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 245, | |
| "step_time": 5.745181904000219 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 20.875, | |
| "completions/mean_terminated_length": 20.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07973140105605125, | |
| "epoch": 1.921875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01369334477931261, | |
| "kl": 0.004745342535898089, | |
| "learning_rate": 1.7491402021185489e-06, | |
| "loss": 4.806690776604228e-05, | |
| "num_tokens": 1184229.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 246, | |
| "step_time": 6.7438787659998525 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.125, | |
| "completions/max_length": 128.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 30.625, | |
| "completions/mean_terminated_length": 16.71428680419922, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09480598196387291, | |
| "epoch": 1.9296875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.04195118695497513, | |
| "kl": 0.008697986835613847, | |
| "learning_rate": 1.7274575140626318e-06, | |
| "loss": 7.138060755096376e-05, | |
| "num_tokens": 1189874.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 247, | |
| "step_time": 15.305213977000221 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.375, | |
| "completions/mean_terminated_length": 13.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12500061094760895, | |
| "epoch": 1.9375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0684664249420166, | |
| "kl": 0.013888376764953136, | |
| "learning_rate": 1.7058388851206187e-06, | |
| "loss": 0.00013920964556746185, | |
| "num_tokens": 1195381.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 248, | |
| "step_time": 5.70922280700006 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 25.375, | |
| "completions/mean_terminated_length": 25.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03869021311402321, | |
| "epoch": 1.9453125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.015580818057060242, | |
| "kl": 0.004994981922209263, | |
| "learning_rate": 1.6842861079061717e-06, | |
| "loss": 4.9939146265387535e-05, | |
| "num_tokens": 1199568.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 249, | |
| "step_time": 6.0302398560002075 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.875, | |
| "completions/mean_terminated_length": 27.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03580416925251484, | |
| "epoch": 1.953125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.012648792937397957, | |
| "kl": 0.004584258887916803, | |
| "learning_rate": 1.6628009695725348e-06, | |
| "loss": 4.5331114961300045e-05, | |
| "num_tokens": 1203687.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 250, | |
| "step_time": 6.181951604999995 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.25, | |
| "completions/mean_terminated_length": 13.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11927280202507973, | |
| "epoch": 1.9609375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.06169552356004715, | |
| "kl": 0.016291253734380007, | |
| "learning_rate": 1.6413852516643468e-06, | |
| "loss": 0.0001629125326871872, | |
| "num_tokens": 1209189.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 251, | |
| "step_time": 5.695720361999975 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 37.0, | |
| "completions/max_terminated_length": 37.0, | |
| "completions/mean_length": 16.125, | |
| "completions/mean_terminated_length": 16.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.10105976462364197, | |
| "epoch": 1.96875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.04228847101330757, | |
| "kl": 0.010480482131242752, | |
| "learning_rate": 1.6200407299699141e-06, | |
| "loss": 9.954295819625258e-05, | |
| "num_tokens": 1214742.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 252, | |
| "step_time": 7.750054464999721 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.04526445455849171, | |
| "epoch": 1.9765625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03174661472439766, | |
| "kl": 0.011470308061689138, | |
| "learning_rate": 1.5987691743739636e-06, | |
| "loss": 0.00011257198639214039, | |
| "num_tokens": 1218788.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 253, | |
| "step_time": 6.203933252000752 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.125, | |
| "completions/mean_terminated_length": 13.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12663870304822922, | |
| "epoch": 1.984375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02504121884703636, | |
| "kl": 0.005651417304761708, | |
| "learning_rate": 1.5775723487108821e-06, | |
| "loss": 5.668877565767616e-05, | |
| "num_tokens": 1224269.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 254, | |
| "step_time": 5.6966200050001135 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.375, | |
| "completions/mean_terminated_length": 19.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06413119472563267, | |
| "epoch": 1.9921875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.036313824355602264, | |
| "kl": 0.00865871855057776, | |
| "learning_rate": 1.5564520106184643e-06, | |
| "loss": 8.569318742956966e-05, | |
| "num_tokens": 1229052.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 255, | |
| "step_time": 6.825095515999692 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 15.0, | |
| "completions/mean_terminated_length": 15.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09427187219262123, | |
| "epoch": 2.0, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.028769060969352722, | |
| "kl": 0.010795064270496368, | |
| "learning_rate": 1.5354099113921614e-06, | |
| "loss": 0.0001041956347762607, | |
| "num_tokens": 1233740.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 256, | |
| "step_time": 6.602496646999953 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 22.25, | |
| "completions/mean_terminated_length": 22.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0705322865396738, | |
| "epoch": 2.0078125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01884527876973152, | |
| "kl": 0.004939868813380599, | |
| "learning_rate": 1.514447795839874e-06, | |
| "loss": 4.6844143071211874e-05, | |
| "num_tokens": 1238474.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 257, | |
| "step_time": 7.474117505999857 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.125, | |
| "completions/mean_terminated_length": 17.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07266378402709961, | |
| "epoch": 2.015625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.022645561024546623, | |
| "kl": 0.0061811115592718124, | |
| "learning_rate": 1.493567402137263e-06, | |
| "loss": 6.193597073433921e-05, | |
| "num_tokens": 1243295.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 258, | |
| "step_time": 6.768364107999787 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 16.0, | |
| "completions/mean_terminated_length": 16.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12317134439945221, | |
| "epoch": 2.0234375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.027730176225304604, | |
| "kl": 0.00469887419603765, | |
| "learning_rate": 1.4727704616836297e-06, | |
| "loss": 4.60884184576571e-05, | |
| "num_tokens": 1248799.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 259, | |
| "step_time": 7.577763272000084 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.25, | |
| "completions/mean_terminated_length": 17.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07146281003952026, | |
| "epoch": 2.03125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.032720040529966354, | |
| "kl": 0.007956057321280241, | |
| "learning_rate": 1.4520586989583406e-06, | |
| "loss": 7.956057379487902e-05, | |
| "num_tokens": 1253513.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 260, | |
| "step_time": 6.917914829999518 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.75, | |
| "completions/mean_terminated_length": 21.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06579671613872051, | |
| "epoch": 2.0390625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02816769853234291, | |
| "kl": 0.008500087773427367, | |
| "learning_rate": 1.431433831377836e-06, | |
| "loss": 7.657032983843237e-05, | |
| "num_tokens": 1258259.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 261, | |
| "step_time": 6.894851472000482 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.25, | |
| "completions/mean_terminated_length": 13.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13017303496599197, | |
| "epoch": 2.046875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.027672773227095604, | |
| "kl": 0.005289856111630797, | |
| "learning_rate": 1.4108975691532273e-06, | |
| "loss": 5.289855835144408e-05, | |
| "num_tokens": 1263765.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 262, | |
| "step_time": 5.753936429000078 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.375, | |
| "completions/mean_terminated_length": 19.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07268621399998665, | |
| "epoch": 2.0546875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007546247914433479, | |
| "kl": 0.002283608540892601, | |
| "learning_rate": 1.3904516151484794e-06, | |
| "loss": 2.186072015319951e-05, | |
| "num_tokens": 1268632.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 263, | |
| "step_time": 6.862648165999872 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 37.0, | |
| "completions/max_terminated_length": 37.0, | |
| "completions/mean_length": 16.0, | |
| "completions/mean_terminated_length": 16.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11497627571225166, | |
| "epoch": 2.0625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.013943173922598362, | |
| "kl": 0.002583169029094279, | |
| "learning_rate": 1.370097664739212e-06, | |
| "loss": 2.6390163839096203e-05, | |
| "num_tokens": 1274160.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 264, | |
| "step_time": 7.761295357000108 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 16.375, | |
| "completions/mean_terminated_length": 16.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11228630319237709, | |
| "epoch": 2.0703125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02251162752509117, | |
| "kl": 0.003555023460648954, | |
| "learning_rate": 1.3498374056721198e-06, | |
| "loss": 3.469496368779801e-05, | |
| "num_tokens": 1279691.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 265, | |
| "step_time": 7.840798843000357 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.0, | |
| "completions/mean_terminated_length": 21.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06668232567608356, | |
| "epoch": 2.078125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005440605338662863, | |
| "kl": 0.0017036688514053822, | |
| "learning_rate": 1.3296725179250274e-06, | |
| "loss": 1.7276026483159512e-05, | |
| "num_tokens": 1284475.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 266, | |
| "step_time": 6.7484774439999455 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 18.125, | |
| "completions/mean_terminated_length": 18.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08780457451939583, | |
| "epoch": 2.0859375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02105136401951313, | |
| "kl": 0.006752714281901717, | |
| "learning_rate": 1.3096046735675795e-06, | |
| "loss": 6.217646296136081e-05, | |
| "num_tokens": 1289212.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 267, | |
| "step_time": 7.876546445999793 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 23.0, | |
| "completions/mean_terminated_length": 23.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03588521108031273, | |
| "epoch": 2.09375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0117345554754138, | |
| "kl": 0.003772490192204714, | |
| "learning_rate": 1.2896355366226e-06, | |
| "loss": 3.7270961911417544e-05, | |
| "num_tokens": 1293364.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 268, | |
| "step_time": 5.872368308999739 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.5, | |
| "completions/mean_terminated_length": 19.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06518647447228432, | |
| "epoch": 2.1015625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.021700672805309296, | |
| "kl": 0.004600343410857022, | |
| "learning_rate": 1.2697667629281025e-06, | |
| "loss": 4.466335667530075e-05, | |
| "num_tokens": 1298152.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 269, | |
| "step_time": 6.83284532000016 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.5, | |
| "completions/mean_terminated_length": 27.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03484191186726093, | |
| "epoch": 2.109375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0065823025070130825, | |
| "kl": 0.002110914676450193, | |
| "learning_rate": 1.2500000000000007e-06, | |
| "loss": 2.08322726393817e-05, | |
| "num_tokens": 1302192.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 270, | |
| "step_time": 5.96173259499983 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 16.25, | |
| "completions/mean_terminated_length": 16.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12080564349889755, | |
| "epoch": 2.1171875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01397998258471489, | |
| "kl": 0.0026146003510802984, | |
| "learning_rate": 1.2303368868954848e-06, | |
| "loss": 2.3293352569453418e-05, | |
| "num_tokens": 1307722.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 271, | |
| "step_time": 7.747071295000296 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 15.125, | |
| "completions/mean_terminated_length": 15.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09482500702142715, | |
| "epoch": 2.125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.014719804748892784, | |
| "kl": 0.0029105256544426084, | |
| "learning_rate": 1.2107790540771208e-06, | |
| "loss": 3.1068855605553836e-05, | |
| "num_tokens": 1312491.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 272, | |
| "step_time": 6.803867777999585 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.0, | |
| "completions/mean_terminated_length": 17.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08690010197460651, | |
| "epoch": 2.1328125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.012722927145659924, | |
| "kl": 0.0026088722806889564, | |
| "learning_rate": 1.1913281232776445e-06, | |
| "loss": 3.0089617212070152e-05, | |
| "num_tokens": 1317183.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 273, | |
| "step_time": 6.73350407099997 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 18.875, | |
| "completions/mean_terminated_length": 18.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.10723384842276573, | |
| "epoch": 2.140625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.012960345484316349, | |
| "kl": 0.001953385421074927, | |
| "learning_rate": 1.1719857073654923e-06, | |
| "loss": 1.958140819624532e-05, | |
| "num_tokens": 1322710.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 274, | |
| "step_time": 7.730757156000436 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 29.0, | |
| "completions/mean_terminated_length": 29.0, | |
| "completions/min_length": 29.0, | |
| "completions/min_terminated_length": 29.0, | |
| "entropy": 0.03018476441502571, | |
| "epoch": 2.1484375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0026580565609037876, | |
| "kl": 0.0016376653802581131, | |
| "learning_rate": 1.1527534102110613e-06, | |
| "loss": 1.6376652638427913e-05, | |
| "num_tokens": 1326802.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 275, | |
| "step_time": 5.899295565000557 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.375, | |
| "completions/mean_terminated_length": 21.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.059426622465252876, | |
| "epoch": 2.15625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.025692613795399666, | |
| "kl": 0.007004954153671861, | |
| "learning_rate": 1.1336328265537195e-06, | |
| "loss": 7.027122774161398e-05, | |
| "num_tokens": 1331677.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 276, | |
| "step_time": 6.831976745000247 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 80.0, | |
| "completions/max_terminated_length": 80.0, | |
| "completions/mean_length": 35.375, | |
| "completions/mean_terminated_length": 35.375, | |
| "completions/min_length": 29.0, | |
| "completions/min_terminated_length": 29.0, | |
| "entropy": 0.03917406965047121, | |
| "epoch": 2.1640625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0032123220153152943, | |
| "kl": 0.001322296040598303, | |
| "learning_rate": 1.1146255418695635e-06, | |
| "loss": 1.3043942090007477e-05, | |
| "num_tokens": 1335820.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 277, | |
| "step_time": 10.114057109999976 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 24.625, | |
| "completions/mean_terminated_length": 24.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.04752049781382084, | |
| "epoch": 2.171875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.010773031041026115, | |
| "kl": 0.004988847824279219, | |
| "learning_rate": 1.0957331322399575e-06, | |
| "loss": 4.370003443909809e-05, | |
| "num_tokens": 1339885.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 278, | |
| "step_time": 5.81339948699997 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 92.0, | |
| "completions/max_terminated_length": 92.0, | |
| "completions/mean_length": 28.875, | |
| "completions/mean_terminated_length": 28.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07032647728919983, | |
| "epoch": 2.1796875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005957606248557568, | |
| "kl": 0.0015352739137597382, | |
| "learning_rate": 1.0769571642208404e-06, | |
| "loss": 1.4340959751280025e-05, | |
| "num_tokens": 1344764.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 279, | |
| "step_time": 11.989179764000255 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 37.0, | |
| "completions/max_terminated_length": 37.0, | |
| "completions/mean_length": 20.0, | |
| "completions/mean_terminated_length": 20.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.15787944197654724, | |
| "epoch": 2.1875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.022516168653964996, | |
| "kl": 0.006525776581838727, | |
| "learning_rate": 1.0582991947128324e-06, | |
| "loss": 6.593053694814444e-05, | |
| "num_tokens": 1348672.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 280, | |
| "step_time": 6.449207413999829 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 17.25, | |
| "completions/mean_terminated_length": 17.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0816731434315443, | |
| "epoch": 2.1953125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.010547587648034096, | |
| "kl": 0.002592157106846571, | |
| "learning_rate": 1.0397607708321302e-06, | |
| "loss": 2.5727105821715668e-05, | |
| "num_tokens": 1353414.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 281, | |
| "step_time": 7.077126720000251 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 25.375, | |
| "completions/mean_terminated_length": 25.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.033613706938922405, | |
| "epoch": 2.203125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0056767649948596954, | |
| "kl": 0.0027879257686436176, | |
| "learning_rate": 1.0213434297822275e-06, | |
| "loss": 2.6398176487418823e-05, | |
| "num_tokens": 1357505.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 282, | |
| "step_time": 5.923250760999508 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 26.875, | |
| "completions/mean_terminated_length": 26.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03769804444164038, | |
| "epoch": 2.2109375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005933691281825304, | |
| "kl": 0.00262093311175704, | |
| "learning_rate": 1.0030486987264436e-06, | |
| "loss": 2.5216926587745547e-05, | |
| "num_tokens": 1361644.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 283, | |
| "step_time": 5.9830027580001115 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 17.875, | |
| "completions/mean_terminated_length": 17.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0936516635119915, | |
| "epoch": 2.21875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.015039868652820587, | |
| "kl": 0.003393694059923291, | |
| "learning_rate": 9.848780946612962e-07, | |
| "loss": 3.5618319088825956e-05, | |
| "num_tokens": 1366347.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 284, | |
| "step_time": 7.200875815000018 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06426311284303665, | |
| "epoch": 2.2265625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007608731277287006, | |
| "kl": 0.001917863031849265, | |
| "learning_rate": 9.66833124290709e-07, | |
| "loss": 1.9463377611828037e-05, | |
| "num_tokens": 1371241.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 285, | |
| "step_time": 6.8967751259997385 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 94.0, | |
| "completions/max_terminated_length": 94.0, | |
| "completions/mean_length": 27.25, | |
| "completions/mean_terminated_length": 27.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06446886621415615, | |
| "epoch": 2.234375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.010835153982043266, | |
| "kl": 0.00325047189835459, | |
| "learning_rate": 9.489152839010799e-07, | |
| "loss": 2.8311722417129204e-05, | |
| "num_tokens": 1376131.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 286, | |
| "step_time": 11.860804265999832 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 17.5, | |
| "completions/mean_terminated_length": 17.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08248795196413994, | |
| "epoch": 2.2421875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.010750790126621723, | |
| "kl": 0.0020258878357708454, | |
| "learning_rate": 9.311260592372045e-07, | |
| "loss": 2.2169147996464744e-05, | |
| "num_tokens": 1380907.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 287, | |
| "step_time": 6.986179221999919 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.5, | |
| "completions/mean_terminated_length": 19.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11641774699091911, | |
| "epoch": 2.25, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.011737891472876072, | |
| "kl": 0.003091278951615095, | |
| "learning_rate": 9.134669253790814e-07, | |
| "loss": 2.849579141184222e-05, | |
| "num_tokens": 1385623.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 288, | |
| "step_time": 6.810695373999806 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.125, | |
| "completions/mean_terminated_length": 21.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06711885333061218, | |
| "epoch": 2.2578125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007158924825489521, | |
| "kl": 0.0025333911762572825, | |
| "learning_rate": 8.959393466195973e-07, | |
| "loss": 2.5275945517932996e-05, | |
| "num_tokens": 1390344.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 289, | |
| "step_time": 6.85957102600014 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 17.25, | |
| "completions/mean_terminated_length": 17.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07853703573346138, | |
| "epoch": 2.265625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01479937881231308, | |
| "kl": 0.004124688799493015, | |
| "learning_rate": 8.785447763431101e-07, | |
| "loss": 4.1246887121815234e-05, | |
| "num_tokens": 1395054.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 290, | |
| "step_time": 6.63891823799986 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.125, | |
| "completions/max_length": 128.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 40.75, | |
| "completions/mean_terminated_length": 28.285715103149414, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0656740702688694, | |
| "epoch": 2.2734375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0060638003051280975, | |
| "kl": 0.0017024054541252553, | |
| "learning_rate": 8.612846569049324e-07, | |
| "loss": 1.726844857330434e-05, | |
| "num_tokens": 1400052.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 291, | |
| "step_time": 14.556338565999795 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 23.375, | |
| "completions/mean_terminated_length": 23.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03545568138360977, | |
| "epoch": 2.28125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007708441931754351, | |
| "kl": 0.0022156074410304427, | |
| "learning_rate": 8.441604195117315e-07, | |
| "loss": 2.2025331418262795e-05, | |
| "num_tokens": 1404051.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 292, | |
| "step_time": 5.789175566999802 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.125, | |
| "completions/mean_terminated_length": 17.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.09139403142035007, | |
| "epoch": 2.2890625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007536021526902914, | |
| "kl": 0.0019620120874606073, | |
| "learning_rate": 8.271734841028553e-07, | |
| "loss": 1.879773844848387e-05, | |
| "num_tokens": 1408800.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 293, | |
| "step_time": 6.592784782999843 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.5, | |
| "completions/mean_terminated_length": 21.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03953526355326176, | |
| "epoch": 2.296875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01195154432207346, | |
| "kl": 0.003599834628403187, | |
| "learning_rate": 8.103252592325897e-07, | |
| "loss": 3.324213685118593e-05, | |
| "num_tokens": 1412752.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 294, | |
| "step_time": 5.7077796439998565 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.125, | |
| "completions/mean_terminated_length": 17.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08125988394021988, | |
| "epoch": 2.3046875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007471000775694847, | |
| "kl": 0.00221208727452904, | |
| "learning_rate": 7.936171419533653e-07, | |
| "loss": 2.2959266061661765e-05, | |
| "num_tokens": 1417441.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 295, | |
| "step_time": 6.576927980000164 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1136050745844841, | |
| "epoch": 2.3125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.011644271202385426, | |
| "kl": 0.0031743868021294475, | |
| "learning_rate": 7.770505176999066e-07, | |
| "loss": 2.477930684108287e-05, | |
| "num_tokens": 1422995.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 296, | |
| "step_time": 7.677933532000225 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.625, | |
| "completions/mean_terminated_length": 19.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07930410280823708, | |
| "epoch": 2.3203125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006496694404631853, | |
| "kl": 0.0013743448653258383, | |
| "learning_rate": 7.606267601743614e-07, | |
| "loss": 1.3506290997611359e-05, | |
| "num_tokens": 1427804.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 297, | |
| "step_time": 6.850105020000228 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1292087510228157, | |
| "epoch": 2.328125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02982700616121292, | |
| "kl": 0.002042026782874018, | |
| "learning_rate": 7.443472312323824e-07, | |
| "loss": 1.8672115402296185e-05, | |
| "num_tokens": 1433332.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 298, | |
| "step_time": 7.489191680000204 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.375, | |
| "completions/mean_terminated_length": 21.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05677143856883049, | |
| "epoch": 2.3359375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007880290038883686, | |
| "kl": 0.0026660520816221833, | |
| "learning_rate": 7.282132807702144e-07, | |
| "loss": 2.6689433070714585e-05, | |
| "num_tokens": 1438211.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 299, | |
| "step_time": 6.8618144679999205 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.5, | |
| "completions/mean_terminated_length": 27.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0277756555005908, | |
| "epoch": 2.34375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0013915153685957193, | |
| "kl": 0.001870743464678526, | |
| "learning_rate": 7.122262466127513e-07, | |
| "loss": 1.8572343833511695e-05, | |
| "num_tokens": 1442435.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 300, | |
| "step_time": 6.021347086999867 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 107.0, | |
| "completions/max_terminated_length": 107.0, | |
| "completions/mean_length": 37.0, | |
| "completions/mean_terminated_length": 37.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05828935466706753, | |
| "epoch": 2.3515625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004984916187822819, | |
| "kl": 0.0012245121179148555, | |
| "learning_rate": 6.963874544026109e-07, | |
| "loss": 1.2662912922678515e-05, | |
| "num_tokens": 1447379.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 301, | |
| "step_time": 13.41408500999978 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 16.25, | |
| "completions/mean_terminated_length": 16.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12101850658655167, | |
| "epoch": 2.359375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.06349971890449524, | |
| "kl": 0.013034343253821135, | |
| "learning_rate": 6.806982174902065e-07, | |
| "loss": 0.00013196206418797374, | |
| "num_tokens": 1452909.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 302, | |
| "step_time": 7.468966810999973 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 29.5, | |
| "completions/mean_terminated_length": 29.5, | |
| "completions/min_length": 29.0, | |
| "completions/min_terminated_length": 29.0, | |
| "entropy": 0.02912551909685135, | |
| "epoch": 2.3671875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0009350188192911446, | |
| "kl": 0.0012789619504474103, | |
| "learning_rate": 6.651598368248494e-07, | |
| "loss": 1.2786756997229531e-05, | |
| "num_tokens": 1456989.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 303, | |
| "step_time": 5.737073586000406 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 108.0, | |
| "completions/max_terminated_length": 108.0, | |
| "completions/mean_length": 28.875, | |
| "completions/mean_terminated_length": 28.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08968518674373627, | |
| "epoch": 2.375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.010912664234638214, | |
| "kl": 0.002259014407172799, | |
| "learning_rate": 6.497736008468703e-07, | |
| "loss": 2.482989293639548e-05, | |
| "num_tokens": 1461908.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 304, | |
| "step_time": 13.122933298000135 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.625, | |
| "completions/mean_terminated_length": 19.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06897314265370369, | |
| "epoch": 2.3828125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0196547731757164, | |
| "kl": 0.006106121756602079, | |
| "learning_rate": 6.345407853807864e-07, | |
| "loss": 5.6373894040007144e-05, | |
| "num_tokens": 1466729.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 305, | |
| "step_time": 7.000840238999899 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.625, | |
| "completions/mean_terminated_length": 13.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13361556828022003, | |
| "epoch": 2.390625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02360568754374981, | |
| "kl": 0.0038382920320145786, | |
| "learning_rate": 6.194626535295059e-07, | |
| "loss": 3.8651305658277124e-05, | |
| "num_tokens": 1472214.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 306, | |
| "step_time": 5.796944548999818 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07942482270300388, | |
| "epoch": 2.3984375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004173581022769213, | |
| "kl": 0.0015490494552068412, | |
| "learning_rate": 6.045404555695935e-07, | |
| "loss": 1.529452310933266e-05, | |
| "num_tokens": 1476946.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 307, | |
| "step_time": 6.881738667000263 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 40.0, | |
| "completions/max_terminated_length": 40.0, | |
| "completions/mean_length": 16.625, | |
| "completions/mean_terminated_length": 16.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11725720018148422, | |
| "epoch": 2.40625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.027200696989893913, | |
| "kl": 0.006251339975278825, | |
| "learning_rate": 5.897754288475979e-07, | |
| "loss": 5.222540858085267e-05, | |
| "num_tokens": 1482503.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 308, | |
| "step_time": 7.865210730999479 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.375, | |
| "completions/mean_terminated_length": 27.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03089694306254387, | |
| "epoch": 2.4140625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.001481092651374638, | |
| "kl": 0.0015604888903908432, | |
| "learning_rate": 5.751687976774523e-07, | |
| "loss": 1.5621018974343315e-05, | |
| "num_tokens": 1486670.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 309, | |
| "step_time": 5.85635403499964 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.25, | |
| "completions/mean_terminated_length": 21.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07554454170167446, | |
| "epoch": 2.421875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005015636794269085, | |
| "kl": 0.001698852691333741, | |
| "learning_rate": 5.607217732389503e-07, | |
| "loss": 1.6451504052383825e-05, | |
| "num_tokens": 1491408.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 310, | |
| "step_time": 6.528806915999667 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 21.75, | |
| "completions/mean_terminated_length": 21.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12106388807296753, | |
| "epoch": 2.4296875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02396441251039505, | |
| "kl": 0.0038617003010585904, | |
| "learning_rate": 5.464355534773217e-07, | |
| "loss": 2.8673672204604372e-05, | |
| "num_tokens": 1496958.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 311, | |
| "step_time": 7.533618949999891 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 18.875, | |
| "completions/mean_terminated_length": 18.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11433938518166542, | |
| "epoch": 2.4375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.003641946241259575, | |
| "kl": 0.001054832770023495, | |
| "learning_rate": 5.323113230038899e-07, | |
| "loss": 9.623323421692476e-06, | |
| "num_tokens": 1502481.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 312, | |
| "step_time": 7.2610713440003565 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 25.0, | |
| "completions/mean_terminated_length": 25.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0287599079310894, | |
| "epoch": 2.4453125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.002698131836950779, | |
| "kl": 0.0020212933886796236, | |
| "learning_rate": 5.183502529978548e-07, | |
| "loss": 2.0212932213325985e-05, | |
| "num_tokens": 1506701.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 313, | |
| "step_time": 5.806462265000391 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 25.375, | |
| "completions/mean_terminated_length": 25.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.032324470579624176, | |
| "epoch": 2.453125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0024203648790717125, | |
| "kl": 0.0017645764746703207, | |
| "learning_rate": 5.045535011091693e-07, | |
| "loss": 1.707239425741136e-05, | |
| "num_tokens": 1510856.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 314, | |
| "step_time": 5.994479979000062 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 17.25, | |
| "completions/mean_terminated_length": 17.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07266517356038094, | |
| "epoch": 2.4609375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.009694607928395271, | |
| "kl": 0.0022006782237440348, | |
| "learning_rate": 4.909222113625545e-07, | |
| "loss": 2.202560062869452e-05, | |
| "num_tokens": 1515614.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 315, | |
| "step_time": 6.553128839000237 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 16.25, | |
| "completions/mean_terminated_length": 16.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1260695531964302, | |
| "epoch": 2.46875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.02785095013678074, | |
| "kl": 0.003913273336365819, | |
| "learning_rate": 4.774575140626317e-07, | |
| "loss": 3.9146947528934106e-05, | |
| "num_tokens": 1521120.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 316, | |
| "step_time": 7.450610239000071 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 29.0, | |
| "completions/mean_terminated_length": 29.0, | |
| "completions/min_length": 29.0, | |
| "completions/min_terminated_length": 29.0, | |
| "entropy": 0.02743313182145357, | |
| "epoch": 2.4765625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0003154528676532209, | |
| "kl": 0.0018573110573925078, | |
| "learning_rate": 4.6416052570020047e-07, | |
| "loss": 1.8573109628050588e-05, | |
| "num_tokens": 1525348.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 317, | |
| "step_time": 6.067128046000107 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 27.5, | |
| "completions/mean_terminated_length": 27.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03275044448673725, | |
| "epoch": 2.484375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0030340480152517557, | |
| "kl": 0.0018103363690897822, | |
| "learning_rate": 4.510323488596588e-07, | |
| "loss": 1.801705002435483e-05, | |
| "num_tokens": 1529396.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 318, | |
| "step_time": 6.102241871999922 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 60.0, | |
| "completions/max_terminated_length": 60.0, | |
| "completions/mean_length": 31.25, | |
| "completions/mean_terminated_length": 31.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.044731512665748596, | |
| "epoch": 2.4921875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006392910145223141, | |
| "kl": 0.002438001334667206, | |
| "learning_rate": 4.380740721275786e-07, | |
| "loss": 2.4828672394505702e-05, | |
| "num_tokens": 1533658.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 319, | |
| "step_time": 8.285051075999945 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 37.0, | |
| "completions/max_terminated_length": 37.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11057645827531815, | |
| "epoch": 2.5, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03642839938402176, | |
| "kl": 0.0035907960700569674, | |
| "learning_rate": 4.252867700024374e-07, | |
| "loss": 4.713074667961337e-05, | |
| "num_tokens": 1539236.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 320, | |
| "step_time": 7.745400238999991 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.875, | |
| "completions/mean_terminated_length": 19.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06698767095804214, | |
| "epoch": 2.5078125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007277606055140495, | |
| "kl": 0.0023683570325374603, | |
| "learning_rate": 4.1267150280552256e-07, | |
| "loss": 2.3156695533543825e-05, | |
| "num_tokens": 1543991.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 321, | |
| "step_time": 7.041696197999954 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.625, | |
| "completions/mean_terminated_length": 13.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12220295518636703, | |
| "epoch": 2.515625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.021708490327000618, | |
| "kl": 0.005996785359457135, | |
| "learning_rate": 4.002293165930088e-07, | |
| "loss": 5.987433178233914e-05, | |
| "num_tokens": 1549476.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 322, | |
| "step_time": 5.8188227460000235 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 21.375, | |
| "completions/mean_terminated_length": 21.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.060368530452251434, | |
| "epoch": 2.5234375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03838368505239487, | |
| "kl": 0.009895860450342298, | |
| "learning_rate": 3.879612430692223e-07, | |
| "loss": 8.399881335208192e-05, | |
| "num_tokens": 1554271.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 323, | |
| "step_time": 6.701541300000372 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.375, | |
| "completions/mean_terminated_length": 19.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07650637812912464, | |
| "epoch": 2.53125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004762283060699701, | |
| "kl": 0.0014340974448714405, | |
| "learning_rate": 3.7586829950108787e-07, | |
| "loss": 1.502779468864901e-05, | |
| "num_tokens": 1558994.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 324, | |
| "step_time": 6.7548362940001425 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 66.0, | |
| "completions/max_terminated_length": 66.0, | |
| "completions/mean_length": 29.375, | |
| "completions/mean_terminated_length": 29.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.2658330798149109, | |
| "epoch": 2.5390625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.003241270314902067, | |
| "kl": 0.0007280520221684128, | |
| "learning_rate": 3.639514886337786e-07, | |
| "loss": 8.224497832998168e-06, | |
| "num_tokens": 1564653.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 325, | |
| "step_time": 10.17069353899933 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07305469736456871, | |
| "epoch": 2.546875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0017727442318573594, | |
| "kl": 0.0009516599820926785, | |
| "learning_rate": 3.5221179860757156e-07, | |
| "loss": 9.93437697616173e-06, | |
| "num_tokens": 1569417.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 326, | |
| "step_time": 6.583777612000176 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 23.0, | |
| "completions/mean_terminated_length": 23.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03469938971102238, | |
| "epoch": 2.5546875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008155560120940208, | |
| "kl": 0.0023509636521339417, | |
| "learning_rate": 3.4065020287590456e-07, | |
| "loss": 2.3022465029498562e-05, | |
| "num_tokens": 1573341.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 327, | |
| "step_time": 5.557240961000389 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 25.0, | |
| "completions/mean_terminated_length": 25.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.029884813353419304, | |
| "epoch": 2.5625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0019974284805357456, | |
| "kl": 0.0016899254987947643, | |
| "learning_rate": 3.292676601246661e-07, | |
| "loss": 1.6899255570024252e-05, | |
| "num_tokens": 1577501.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 328, | |
| "step_time": 5.841788397000528 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 23.25, | |
| "completions/mean_terminated_length": 23.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03349179029464722, | |
| "epoch": 2.5703125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006807905156165361, | |
| "kl": 0.002164472476579249, | |
| "learning_rate": 3.18065114192693e-07, | |
| "loss": 2.147164923371747e-05, | |
| "num_tokens": 1581531.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 329, | |
| "step_time": 5.71764247999954 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 20.75, | |
| "completions/mean_terminated_length": 20.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.10336236655712128, | |
| "epoch": 2.578125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.013856982812285423, | |
| "kl": 0.003586582955904305, | |
| "learning_rate": 3.0704349399351437e-07, | |
| "loss": 3.917415233445354e-05, | |
| "num_tokens": 1586265.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 330, | |
| "step_time": 6.691765174000011 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 25.125, | |
| "completions/mean_terminated_length": 25.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0626723226159811, | |
| "epoch": 2.5859375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004872177727520466, | |
| "kl": 0.0017109083128161728, | |
| "learning_rate": 2.962037134383211e-07, | |
| "loss": 1.6971382137853652e-05, | |
| "num_tokens": 1591170.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 331, | |
| "step_time": 7.28137495899955 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.375, | |
| "completions/mean_terminated_length": 13.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1446835845708847, | |
| "epoch": 2.59375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.03924147039651871, | |
| "kl": 0.0011937321105506271, | |
| "learning_rate": 2.855466713601868e-07, | |
| "loss": 1.1964344594161958e-05, | |
| "num_tokens": 1596649.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 332, | |
| "step_time": 5.465260511999986 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 25.25, | |
| "completions/mean_terminated_length": 25.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05687732808291912, | |
| "epoch": 2.6015625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0024226743262261152, | |
| "kl": 0.001208013971336186, | |
| "learning_rate": 2.750732514395363e-07, | |
| "loss": 1.1762923350033816e-05, | |
| "num_tokens": 1601571.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 333, | |
| "step_time": 7.622067566999704 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.5, | |
| "completions/mean_terminated_length": 19.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0665333904325962, | |
| "epoch": 2.609375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.009631386026740074, | |
| "kl": 0.002712785266339779, | |
| "learning_rate": 2.647843221308721e-07, | |
| "loss": 2.844407754309941e-05, | |
| "num_tokens": 1606451.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 334, | |
| "step_time": 6.9531723879999845 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 77.0, | |
| "completions/max_terminated_length": 77.0, | |
| "completions/mean_length": 23.875, | |
| "completions/mean_terminated_length": 23.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.371206421405077, | |
| "epoch": 2.6171875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00424219761043787, | |
| "kl": 0.0011196635314263403, | |
| "learning_rate": 2.5468073659076e-07, | |
| "loss": 1.2226804756210186e-05, | |
| "num_tokens": 1612018.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 335, | |
| "step_time": 10.750021371999992 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06773288920521736, | |
| "epoch": 2.625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.001815224764868617, | |
| "kl": 0.001531377958599478, | |
| "learning_rate": 2.44763332607087e-07, | |
| "loss": 1.508441346231848e-05, | |
| "num_tokens": 1616870.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 336, | |
| "step_time": 7.147886858999755 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 27.25, | |
| "completions/mean_terminated_length": 27.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05348302982747555, | |
| "epoch": 2.6328125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.002274098340421915, | |
| "kl": 0.0012252243468537927, | |
| "learning_rate": 2.3503293252959136e-07, | |
| "loss": 1.2252243323018774e-05, | |
| "num_tokens": 1621724.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 337, | |
| "step_time": 7.523159131000284 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 39.0, | |
| "completions/max_terminated_length": 39.0, | |
| "completions/mean_length": 23.625, | |
| "completions/mean_terminated_length": 23.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1177109070122242, | |
| "epoch": 2.640625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.011636418290436268, | |
| "kl": 0.003375065280124545, | |
| "learning_rate": 2.2549034320167501e-07, | |
| "loss": 3.056726563954726e-05, | |
| "num_tokens": 1626517.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 338, | |
| "step_time": 7.475345586000003 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07587442919611931, | |
| "epoch": 2.6484375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.003583204001188278, | |
| "kl": 0.001523865619674325, | |
| "learning_rate": 2.1613635589349756e-07, | |
| "loss": 1.486748533352511e-05, | |
| "num_tokens": 1631389.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 339, | |
| "step_time": 6.883027566999772 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 27.0, | |
| "completions/mean_terminated_length": 27.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.02992851845920086, | |
| "epoch": 2.65625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0010637313826009631, | |
| "kl": 0.0017816239269450307, | |
| "learning_rate": 2.0697174623636795e-07, | |
| "loss": 1.7696058421279304e-05, | |
| "num_tokens": 1635621.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 340, | |
| "step_time": 5.765514154000357 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 40.0, | |
| "completions/max_terminated_length": 40.0, | |
| "completions/mean_length": 24.375, | |
| "completions/mean_terminated_length": 24.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06319654919207096, | |
| "epoch": 2.6640625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004237866494804621, | |
| "kl": 0.0019926169188693166, | |
| "learning_rate": 1.9799727415842323e-07, | |
| "loss": 2.3374921511276625e-05, | |
| "num_tokens": 1640408.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 341, | |
| "step_time": 7.587094982999588 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 16.125, | |
| "completions/mean_terminated_length": 16.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12324276193976402, | |
| "epoch": 2.671875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004930234979838133, | |
| "kl": 0.0005672876577591524, | |
| "learning_rate": 1.8921368382162352e-07, | |
| "loss": 6.185021447890904e-06, | |
| "num_tokens": 1645961.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 342, | |
| "step_time": 7.826221096000154 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 115.0, | |
| "completions/max_terminated_length": 115.0, | |
| "completions/mean_length": 34.25, | |
| "completions/mean_terminated_length": 34.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0991355199366808, | |
| "epoch": 2.6796875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0027120131999254227, | |
| "kl": 0.0009974082058761269, | |
| "learning_rate": 1.8062170356003854e-07, | |
| "loss": 9.134013453149237e-06, | |
| "num_tokens": 1650807.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 343, | |
| "step_time": 13.32459857799995 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.875, | |
| "completions/mean_terminated_length": 21.875, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07806593738496304, | |
| "epoch": 2.6875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0022448606323450804, | |
| "kl": 0.0018360121175646782, | |
| "learning_rate": 1.7222204581946038e-07, | |
| "loss": 1.7249569282284938e-05, | |
| "num_tokens": 1655702.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 344, | |
| "step_time": 6.476274790999923 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 22.0, | |
| "completions/mean_terminated_length": 22.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.0640562754124403, | |
| "epoch": 2.6953125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.013984336517751217, | |
| "kl": 0.002910680166678503, | |
| "learning_rate": 1.6401540709832242e-07, | |
| "loss": 3.365988959558308e-05, | |
| "num_tokens": 1660514.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 345, | |
| "step_time": 7.252558954999586 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 82.0, | |
| "completions/max_terminated_length": 82.0, | |
| "completions/mean_length": 33.625, | |
| "completions/mean_terminated_length": 33.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03660286124795675, | |
| "epoch": 2.703125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.003199823433533311, | |
| "kl": 0.0017720660544000566, | |
| "learning_rate": 1.5600246788994938e-07, | |
| "loss": 1.750032060954254e-05, | |
| "num_tokens": 1664703.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 346, | |
| "step_time": 9.787063931000375 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 19.625, | |
| "completions/mean_terminated_length": 19.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06424449197947979, | |
| "epoch": 2.7109375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0034813072998076677, | |
| "kl": 0.0016274115769192576, | |
| "learning_rate": 1.4818389262612948e-07, | |
| "loss": 1.6602698451606557e-05, | |
| "num_tokens": 1669584.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 347, | |
| "step_time": 6.833064813999499 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 27.25, | |
| "completions/mean_terminated_length": 27.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06542891636490822, | |
| "epoch": 2.71875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0009135848376899958, | |
| "kl": 0.0008284190844278783, | |
| "learning_rate": 1.4056032962202038e-07, | |
| "loss": 8.49508069222793e-06, | |
| "num_tokens": 1674430.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 348, | |
| "step_time": 7.179126660000293 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.625, | |
| "completions/mean_terminated_length": 13.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1259615123271942, | |
| "epoch": 2.7265625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006351651158183813, | |
| "kl": 0.0011312126298435032, | |
| "learning_rate": 1.3313241102239056e-07, | |
| "loss": 1.1335807357681915e-05, | |
| "num_tokens": 1679939.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 349, | |
| "step_time": 5.610404259999996 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 23.0, | |
| "completions/mean_terminated_length": 23.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.032224283553659916, | |
| "epoch": 2.734375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0039573912508785725, | |
| "kl": 0.0017619336140342057, | |
| "learning_rate": 1.2590075274920206e-07, | |
| "loss": 1.7461547031416558e-05, | |
| "num_tokens": 1684007.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 350, | |
| "step_time": 5.557787984000242 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.5, | |
| "completions/mean_terminated_length": 21.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06698081642389297, | |
| "epoch": 2.7421875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.002890112577006221, | |
| "kl": 0.0011417069763410836, | |
| "learning_rate": 1.1886595445053745e-07, | |
| "loss": 1.179358150693588e-05, | |
| "num_tokens": 1688915.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 351, | |
| "step_time": 6.857757833000051 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 21.625, | |
| "completions/mean_terminated_length": 21.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.05869276635348797, | |
| "epoch": 2.75, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0013901939382776618, | |
| "kl": 0.0012589980033226311, | |
| "learning_rate": 1.120285994508799e-07, | |
| "loss": 1.2590622645802796e-05, | |
| "num_tokens": 1693788.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 352, | |
| "step_time": 6.611652337999658 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 37.0, | |
| "completions/max_terminated_length": 37.0, | |
| "completions/mean_length": 20.375, | |
| "completions/mean_terminated_length": 20.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07492958009243011, | |
| "epoch": 2.7578125, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 2.069866895675659, | |
| "kl": 0.37729693634901196, | |
| "learning_rate": 1.053892547027402e-07, | |
| "loss": -0.09150571376085281, | |
| "num_tokens": 1698687.0, | |
| "reward": 0.887499988079071, | |
| "reward_std": 0.3181980550289154, | |
| "rewards/reward_fn/mean": 0.887499988079071, | |
| "rewards/reward_fn/std": 0.3181980550289154, | |
| "step": 353, | |
| "step_time": 7.622229282999797 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.125, | |
| "completions/mean_terminated_length": 19.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07208308577537537, | |
| "epoch": 2.765625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.001533597824163735, | |
| "kl": 0.0011728731915354729, | |
| "learning_rate": 9.894847073964875e-08, | |
| "loss": 1.1982096111751162e-05, | |
| "num_tokens": 1703472.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 354, | |
| "step_time": 6.501406307999787 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.375, | |
| "completions/mean_terminated_length": 13.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13054991513490677, | |
| "epoch": 2.7734375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.008911830373108387, | |
| "kl": 0.0018170861876569688, | |
| "learning_rate": 9.270678163050218e-08, | |
| "loss": 1.823622551455628e-05, | |
| "num_tokens": 1708979.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 355, | |
| "step_time": 5.554297769000186 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 23.0, | |
| "completions/mean_terminated_length": 23.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.036714909598231316, | |
| "epoch": 2.78125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006838838569819927, | |
| "kl": 0.002757440961431712, | |
| "learning_rate": 8.666470493528007e-08, | |
| "loss": 2.4345088604604825e-05, | |
| "num_tokens": 1712979.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 356, | |
| "step_time": 5.553534669000328 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 16.125, | |
| "completions/mean_terminated_length": 16.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12004699558019638, | |
| "epoch": 2.7890625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.003003190504387021, | |
| "kl": 0.0006363969732774422, | |
| "learning_rate": 8.082274166213016e-08, | |
| "loss": 6.007579941069707e-06, | |
| "num_tokens": 1718508.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 357, | |
| "step_time": 7.342511635000392 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 27.0, | |
| "completions/mean_terminated_length": 27.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.031998704187572, | |
| "epoch": 2.796875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0018181306077167392, | |
| "kl": 0.0013061033096164465, | |
| "learning_rate": 7.518137622582189e-08, | |
| "loss": 1.2957580111105926e-05, | |
| "num_tokens": 1722584.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 358, | |
| "step_time": 5.5584459810002045 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 25.0, | |
| "completions/mean_terminated_length": 25.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03392230533063412, | |
| "epoch": 2.8046875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005436773411929607, | |
| "kl": 0.002172080159652978, | |
| "learning_rate": 6.974107640758176e-08, | |
| "loss": 2.063044303213246e-05, | |
| "num_tokens": 1726536.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 359, | |
| "step_time": 5.491190248999828 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 16.125, | |
| "completions/mean_terminated_length": 16.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12706465646624565, | |
| "epoch": 2.8125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.005568970926105976, | |
| "kl": 0.0011207733186893165, | |
| "learning_rate": 6.450229331630253e-08, | |
| "loss": 1.0281852155458182e-05, | |
| "num_tokens": 1732041.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 360, | |
| "step_time": 7.323580845000379 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 27.0, | |
| "completions/mean_terminated_length": 27.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.028290102258324623, | |
| "epoch": 2.8203125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0013824108755216002, | |
| "kl": 0.0015246181283146143, | |
| "learning_rate": 5.946546135113862e-08, | |
| "loss": 1.5194093066384085e-05, | |
| "num_tokens": 1736217.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 361, | |
| "step_time": 5.700615629000367 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 26.75, | |
| "completions/mean_terminated_length": 26.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06074472516775131, | |
| "epoch": 2.828125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.003173437202349305, | |
| "kl": 0.0010228622995782644, | |
| "learning_rate": 5.463099816548578e-08, | |
| "loss": 1.052904008247424e-05, | |
| "num_tokens": 1741115.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 362, | |
| "step_time": 7.050874709999789 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 40.0, | |
| "completions/max_terminated_length": 40.0, | |
| "completions/mean_length": 20.375, | |
| "completions/mean_terminated_length": 20.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07005861215293407, | |
| "epoch": 2.8359375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0047007217071950436, | |
| "kl": 0.001660762238316238, | |
| "learning_rate": 4.999930463234964e-08, | |
| "loss": 1.6024239812395535e-05, | |
| "num_tokens": 1746014.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 363, | |
| "step_time": 7.63879414999974 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 16.125, | |
| "completions/mean_terminated_length": 16.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12410368397831917, | |
| "epoch": 2.84375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.011131868697702885, | |
| "kl": 0.0020194172975607216, | |
| "learning_rate": 4.557076481110367e-08, | |
| "loss": 2.2339860151987523e-05, | |
| "num_tokens": 1751543.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 364, | |
| "step_time": 7.354793092999898 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.125, | |
| "completions/mean_terminated_length": 13.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.1403006836771965, | |
| "epoch": 2.8515625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00545878428965807, | |
| "kl": 0.0008273630810435861, | |
| "learning_rate": 4.134574591564494e-08, | |
| "loss": 8.271908882306889e-06, | |
| "num_tokens": 1757048.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 365, | |
| "step_time": 5.55601100199965 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 23.125, | |
| "completions/mean_terminated_length": 23.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07144136913120747, | |
| "epoch": 2.859375, | |
| "frac_reward_zero_std": 0.5, | |
| "grad_norm": 0.8153831958770752, | |
| "kl": 0.6430793326580897, | |
| "learning_rate": 3.732459828394402e-08, | |
| "loss": -0.04161795973777771, | |
| "num_tokens": 1761901.0, | |
| "reward": 0.8812500238418579, | |
| "reward_std": 0.3358757197856903, | |
| "rewards/reward_fn/mean": 0.8812500238418579, | |
| "rewards/reward_fn/std": 0.3358757197856903, | |
| "step": 366, | |
| "step_time": 7.062660026999765 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 25.125, | |
| "completions/mean_terminated_length": 25.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06979860179126263, | |
| "epoch": 2.8671875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004871731158345938, | |
| "kl": 0.0016156822093762457, | |
| "learning_rate": 3.3507655348995194e-08, | |
| "loss": 1.5751948012621142e-05, | |
| "num_tokens": 1766674.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 367, | |
| "step_time": 7.081914410000536 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 19.25, | |
| "completions/mean_terminated_length": 19.25, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.10976235195994377, | |
| "epoch": 2.875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004110577516257763, | |
| "kl": 0.0005710943078156561, | |
| "learning_rate": 2.98952336111677e-08, | |
| "loss": 6.769578249077313e-06, | |
| "num_tokens": 1772252.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 368, | |
| "step_time": 7.688176613999985 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 84.0, | |
| "completions/max_terminated_length": 84.0, | |
| "completions/mean_length": 34.625, | |
| "completions/mean_terminated_length": 34.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.04085162188857794, | |
| "epoch": 2.8828125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.002521694405004382, | |
| "kl": 0.0012893079547211528, | |
| "learning_rate": 2.6487632611962578e-08, | |
| "loss": 1.2892131053376943e-05, | |
| "num_tokens": 1776517.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 369, | |
| "step_time": 9.961404026999844 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.08238299190998077, | |
| "epoch": 2.890625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007919838652014732, | |
| "kl": 0.002668961475137621, | |
| "learning_rate": 2.3285134909173113e-08, | |
| "loss": 2.5326695322291926e-05, | |
| "num_tokens": 1781345.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 370, | |
| "step_time": 6.4334861359998285 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 40.0, | |
| "completions/max_terminated_length": 40.0, | |
| "completions/mean_length": 16.5, | |
| "completions/mean_terminated_length": 16.5, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12997304648160934, | |
| "epoch": 2.8984375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00875998567789793, | |
| "kl": 0.0007102387316990644, | |
| "learning_rate": 2.028800605345771e-08, | |
| "loss": 6.7351170400797855e-06, | |
| "num_tokens": 1786897.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 371, | |
| "step_time": 7.530595849999827 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07069241255521774, | |
| "epoch": 2.90625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0031238263472914696, | |
| "kl": 0.0017178311245515943, | |
| "learning_rate": 1.7496494566317247e-08, | |
| "loss": 1.785397034836933e-05, | |
| "num_tokens": 1791725.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 372, | |
| "step_time": 6.554121677999774 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 16.125, | |
| "completions/mean_terminated_length": 16.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.11858610808849335, | |
| "epoch": 2.9140625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.01251543965190649, | |
| "kl": 0.0027800038806162775, | |
| "learning_rate": 1.4910831919490997e-08, | |
| "loss": 2.9193033697083592e-05, | |
| "num_tokens": 1797254.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 373, | |
| "step_time": 7.373935341000106 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 36.0, | |
| "completions/max_terminated_length": 36.0, | |
| "completions/mean_length": 16.375, | |
| "completions/mean_terminated_length": 16.375, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12266703695058823, | |
| "epoch": 2.921875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.04878854751586914, | |
| "kl": 0.011298867233563215, | |
| "learning_rate": 1.2531232515760328e-08, | |
| "loss": 9.663843229645863e-05, | |
| "num_tokens": 1802761.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 374, | |
| "step_time": 7.402829858999667 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 30.0, | |
| "completions/max_terminated_length": 30.0, | |
| "completions/mean_length": 25.75, | |
| "completions/mean_terminated_length": 25.75, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.03546947054564953, | |
| "epoch": 2.9296875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006199007388204336, | |
| "kl": 0.001919182192068547, | |
| "learning_rate": 1.0357893671171793e-08, | |
| "loss": 1.9191820683772676e-05, | |
| "num_tokens": 1806727.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 375, | |
| "step_time": 5.591665662999731 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 27.0, | |
| "completions/mean_terminated_length": 27.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.030339499935507774, | |
| "epoch": 2.9375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.001278455718420446, | |
| "kl": 0.001506837084889412, | |
| "learning_rate": 8.390995598676067e-09, | |
| "loss": 1.4967106835683808e-05, | |
| "num_tokens": 1810895.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 376, | |
| "step_time": 5.731536610000148 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 81.0, | |
| "completions/max_terminated_length": 81.0, | |
| "completions/mean_length": 35.5, | |
| "completions/mean_terminated_length": 35.5, | |
| "completions/min_length": 29.0, | |
| "completions/min_terminated_length": 29.0, | |
| "entropy": 0.03009831439703703, | |
| "epoch": 2.9453125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006957578472793102, | |
| "kl": 0.001761360326781869, | |
| "learning_rate": 6.63070139318378e-09, | |
| "loss": 1.8286591512151062e-05, | |
| "num_tokens": 1815119.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 377, | |
| "step_time": 9.591905867999685 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 19.0, | |
| "completions/mean_terminated_length": 19.0, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.07046742364764214, | |
| "epoch": 2.953125, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.007482482586055994, | |
| "kl": 0.0023475169437006116, | |
| "learning_rate": 5.077157018041623e-09, | |
| "loss": 2.212448998761829e-05, | |
| "num_tokens": 1819847.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 378, | |
| "step_time": 6.737704846999804 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 14.0, | |
| "completions/mean_terminated_length": 14.0, | |
| "completions/min_length": 14.0, | |
| "completions/min_terminated_length": 14.0, | |
| "entropy": 0.12937615811824799, | |
| "epoch": 2.9609375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.023704534396529198, | |
| "kl": 0.004825950600206852, | |
| "learning_rate": 3.730491292930072e-09, | |
| "loss": 4.825950600206852e-05, | |
| "num_tokens": 1825355.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 379, | |
| "step_time": 5.74336707700013 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 14.0, | |
| "completions/max_terminated_length": 14.0, | |
| "completions/mean_length": 13.125, | |
| "completions/mean_terminated_length": 13.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.13150636106729507, | |
| "epoch": 2.96875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.004724206868559122, | |
| "kl": 0.0005312660941854119, | |
| "learning_rate": 2.590815883181108e-09, | |
| "loss": 5.310364940669388e-06, | |
| "num_tokens": 1830860.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 380, | |
| "step_time": 5.56844295000019 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 16.125, | |
| "completions/mean_terminated_length": 16.125, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.12551702931523323, | |
| "epoch": 2.9765625, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0012423633597791195, | |
| "kl": 0.00047561304381815717, | |
| "learning_rate": 1.6582252905186779e-09, | |
| "loss": 5.338163646229077e-06, | |
| "num_tokens": 1836389.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 381, | |
| "step_time": 7.517625181999847 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 38.0, | |
| "completions/max_terminated_length": 38.0, | |
| "completions/mean_length": 25.625, | |
| "completions/mean_terminated_length": 25.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06911934912204742, | |
| "epoch": 2.984375, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.006786187645047903, | |
| "kl": 0.0018059620633721352, | |
| "learning_rate": 9.32796845223294e-10, | |
| "loss": 1.838826938183047e-05, | |
| "num_tokens": 1841294.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 382, | |
| "step_time": 7.187780472999748 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 102.0, | |
| "completions/max_terminated_length": 102.0, | |
| "completions/mean_length": 28.625, | |
| "completions/mean_terminated_length": 28.625, | |
| "completions/min_length": 13.0, | |
| "completions/min_terminated_length": 13.0, | |
| "entropy": 0.06470359116792679, | |
| "epoch": 2.9921875, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.00552311772480607, | |
| "kl": 0.0018424552981741726, | |
| "learning_rate": 4.1459069971938604e-10, | |
| "loss": 1.6110756405396387e-05, | |
| "num_tokens": 1846227.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 383, | |
| "step_time": 12.270986474999972 | |
| }, | |
| { | |
| "clip_ratio/high_max": 0.0, | |
| "clip_ratio/high_mean": 0.0, | |
| "clip_ratio/low_mean": 0.0, | |
| "clip_ratio/low_min": 0.0, | |
| "clip_ratio/region_mean": 0.0, | |
| "completions/clipped_ratio": 0.0, | |
| "completions/max_length": 29.0, | |
| "completions/max_terminated_length": 29.0, | |
| "completions/mean_length": 29.0, | |
| "completions/mean_terminated_length": 29.0, | |
| "completions/min_length": 29.0, | |
| "completions/min_terminated_length": 29.0, | |
| "entropy": 0.028123882599174976, | |
| "epoch": 3.0, | |
| "frac_reward_zero_std": 1.0, | |
| "grad_norm": 0.0005120371351949871, | |
| "kl": 0.0014926688745617867, | |
| "learning_rate": 1.0364982358707087e-10, | |
| "loss": 1.4926688891137019e-05, | |
| "num_tokens": 1850395.0, | |
| "reward": 1.0, | |
| "reward_std": 0.0, | |
| "rewards/reward_fn/mean": 1.0, | |
| "rewards/reward_fn/std": 0.0, | |
| "step": 384, | |
| "step_time": 5.804545400000279 | |
| } | |
| ], | |
| "logging_steps": 1, | |
| "max_steps": 384, | |
| "num_input_tokens_seen": 1850395, | |
| "num_train_epochs": 3, | |
| "save_steps": 500, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 0.0, | |
| "train_batch_size": 4, | |
| "trial_name": null, | |
| "trial_params": null | |
| } |