IncidentResponseDetective / data /trainer_state.json
90shikhar08's picture
Add trainer_state.json and download script for training provenance
a5ca2f8
Raw
History Blame Contribute Delete
368 kB
Invalid JSON:Unexpected token 'N', ..."ad_norm": NaN, "... is not valid JSON
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 3.0,
"eval_steps": 500,
"global_step": 384,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 23.625,
"completions/mean_terminated_length": 23.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09965698793530464,
"epoch": 0.0078125,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.13743257522583,
"kl": 9.029518110992285e-08,
"learning_rate": 0.0,
"loss": -0.06901008635759354,
"num_tokens": 4097.0,
"reward": 0.768750011920929,
"reward_std": 0.4300643801689148,
"rewards/reward_fn/mean": 0.768750011920929,
"rewards/reward_fn/std": 0.4300643801689148,
"step": 1,
"step_time": 7.910418492999952
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 19.625,
"completions/mean_terminated_length": 19.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13235441222786903,
"epoch": 0.015625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0,
"kl": 0.0,
"learning_rate": 1.282051282051282e-07,
"loss": 0.0,
"num_tokens": 9674.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 2,
"step_time": 7.016720726000017
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 22.5,
"completions/mean_terminated_length": 22.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11623429134488106,
"epoch": 0.0234375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0009358773822896183,
"kl": 8.32239061310247e-06,
"learning_rate": 2.564102564102564e-07,
"loss": 8.256898098579768e-08,
"num_tokens": 14558.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 3,
"step_time": 6.771791392000068
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 23.5,
"completions/mean_terminated_length": 23.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09514202550053596,
"epoch": 0.03125,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.5667312145233154,
"kl": 3.819915718850098e-06,
"learning_rate": 3.846153846153847e-07,
"loss": -0.08508844673633575,
"num_tokens": 18626.0,
"reward": 0.8812500238418579,
"reward_std": 0.3358757197856903,
"rewards/reward_fn/mean": 0.8812500238418579,
"rewards/reward_fn/std": 0.3358757197856903,
"step": 4,
"step_time": 5.745465063999859
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.14030911028385162,
"epoch": 0.0390625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0005072976928204298,
"kl": 4.893604227618198e-06,
"learning_rate": 5.128205128205128e-07,
"loss": 4.175808498985134e-08,
"num_tokens": 24180.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 5,
"step_time": 7.145598770999982
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.375,
"completions/mean_terminated_length": 21.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11003290489315987,
"epoch": 0.046875,
"frac_reward_zero_std": 0.5,
"grad_norm": 3.21610426902771,
"kl": 1.439371067135653e-05,
"learning_rate": 6.41025641025641e-07,
"loss": 0.008770093321800232,
"num_tokens": 28935.0,
"reward": 0.887499988079071,
"reward_std": 0.3181980550289154,
"rewards/reward_fn/mean": 0.887499988079071,
"rewards/reward_fn/std": 0.3181980550289154,
"step": 6,
"step_time": 6.531211202000009
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.625,
"completions/mean_terminated_length": 21.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12766291946172714,
"epoch": 0.0546875,
"frac_reward_zero_std": 0.0,
"grad_norm": 2.585819959640503,
"kl": 2.266431465614005e-06,
"learning_rate": 7.692307692307694e-07,
"loss": -0.25185173749923706,
"num_tokens": 32988.0,
"reward": 0.643750011920929,
"reward_std": 0.49384605884552,
"rewards/reward_fn/mean": 0.643750011920929,
"rewards/reward_fn/std": 0.49384605884552,
"step": 7,
"step_time": 5.520926363000058
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.75,
"completions/mean_terminated_length": 21.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11385262385010719,
"epoch": 0.0625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00018979530432261527,
"kl": 3.1351814868685324e-06,
"learning_rate": 8.974358974358975e-07,
"loss": 3.0823137819879776e-08,
"num_tokens": 37866.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 8,
"step_time": 6.259193151999966
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.25,
"completions/mean_terminated_length": 13.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1592414677143097,
"epoch": 0.0703125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0010657254606485367,
"kl": 1.2110989018765395e-05,
"learning_rate": 1.0256410256410257e-06,
"loss": 1.211098918929565e-07,
"num_tokens": 43344.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 9,
"step_time": 4.96420772700003
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1171259917318821,
"epoch": 0.078125,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.561084508895874,
"kl": 7.743469495835598e-06,
"learning_rate": 1.153846153846154e-06,
"loss": -0.05262097343802452,
"num_tokens": 47400.0,
"reward": 0.875,
"reward_std": 0.3535533845424652,
"rewards/reward_fn/mean": 0.875,
"rewards/reward_fn/std": 0.3535533845424652,
"step": 10,
"step_time": 5.405768857999988
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 13.0,
"completions/max_terminated_length": 13.0,
"completions/mean_length": 13.0,
"completions/mean_terminated_length": 13.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.16070035845041275,
"epoch": 0.0859375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00039809694862924516,
"kl": 4.659478577195841e-06,
"learning_rate": 1.282051282051282e-06,
"loss": 4.659478491930713e-08,
"num_tokens": 52904.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 11,
"step_time": 5.233364768000001
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 23.25,
"completions/mean_terminated_length": 23.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11492524296045303,
"epoch": 0.09375,
"frac_reward_zero_std": 0.0,
"grad_norm": NaN,
"kl": 8.350619191332953e-05,
"learning_rate": 1.4102564102564104e-06,
"loss": -0.1988813430070877,
"num_tokens": 57022.0,
"reward": 0.7687499523162842,
"reward_std": 0.4284002482891083,
"rewards/reward_fn/mean": 0.7687499523162842,
"rewards/reward_fn/std": 0.4284002482891083,
"step": 12,
"step_time": 5.546596589000046
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.125,
"completions/mean_terminated_length": 19.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.17772966623306274,
"epoch": 0.1015625,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.7892787456512451,
"kl": 1.1041468042094493e-05,
"learning_rate": 1.5384615384615387e-06,
"loss": -0.147029310464859,
"num_tokens": 61727.0,
"reward": 0.875,
"reward_std": 0.3535533845424652,
"rewards/reward_fn/mean": 0.875,
"rewards/reward_fn/std": 0.3535533845424652,
"step": 13,
"step_time": 6.295519733999981
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.0,
"completions/mean_terminated_length": 21.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13657044991850853,
"epoch": 0.109375,
"frac_reward_zero_std": 0.5,
"grad_norm": 2.4697511196136475,
"kl": 0.00011959002404182684,
"learning_rate": 1.6666666666666667e-06,
"loss": -0.07215305417776108,
"num_tokens": 65771.0,
"reward": 0.7875000238418579,
"reward_std": 0.3934735357761383,
"rewards/reward_fn/mean": 0.7875000238418579,
"rewards/reward_fn/std": 0.3934735655784607,
"step": 14,
"step_time": 5.657076707999977
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 66.0,
"completions/max_terminated_length": 66.0,
"completions/mean_length": 29.0,
"completions/mean_terminated_length": 29.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.10727399960160255,
"epoch": 0.1171875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0006498436559922993,
"kl": 2.7017442334908992e-05,
"learning_rate": 1.794871794871795e-06,
"loss": 2.6791002483150805e-07,
"num_tokens": 70595.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 15,
"step_time": 9.642551118000029
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 22.0,
"completions/mean_terminated_length": 22.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09909634664654732,
"epoch": 0.125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0024827050510793924,
"kl": 6.381553976098076e-05,
"learning_rate": 1.9230769230769234e-06,
"loss": 6.585330538655398e-07,
"num_tokens": 75331.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 16,
"step_time": 7.090756369000019
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.625,
"completions/mean_terminated_length": 19.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11970487236976624,
"epoch": 0.1328125,
"frac_reward_zero_std": 0.0,
"grad_norm": 2.5904784202575684,
"kl": 0.0007240207705763169,
"learning_rate": 2.0512820512820513e-06,
"loss": -0.2588059902191162,
"num_tokens": 79512.0,
"reward": 0.5437500476837158,
"reward_std": 0.49021676182746887,
"rewards/reward_fn/mean": 0.5437500476837158,
"rewards/reward_fn/std": 0.49021679162979126,
"step": 17,
"step_time": 5.971762764999994
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.125,
"completions/mean_terminated_length": 19.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.10251206904649734,
"epoch": 0.140625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.003479904029518366,
"kl": 0.00019486098608467728,
"learning_rate": 2.1794871794871797e-06,
"loss": 1.9753097149077803e-06,
"num_tokens": 84289.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 18,
"step_time": 6.534459332000097
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 40.0,
"completions/max_terminated_length": 40.0,
"completions/mean_length": 16.875,
"completions/mean_terminated_length": 16.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13044045120477676,
"epoch": 0.1484375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006375588476657867,
"kl": 0.00021761858806712553,
"learning_rate": 2.307692307692308e-06,
"loss": 2.2181316126079764e-06,
"num_tokens": 89848.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 19,
"step_time": 8.074243104000061
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 25.125,
"completions/mean_terminated_length": 25.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07144641503691673,
"epoch": 0.15625,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.383427381515503,
"kl": 0.00042095681419596076,
"learning_rate": 2.435897435897436e-06,
"loss": -0.111912801861763,
"num_tokens": 93977.0,
"reward": 0.8812500238418579,
"reward_std": 0.3358757197856903,
"rewards/reward_fn/mean": 0.8812500238418579,
"rewards/reward_fn/std": 0.3358757197856903,
"step": 20,
"step_time": 5.788662745999886
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.375,
"completions/mean_terminated_length": 19.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.10241465270519257,
"epoch": 0.1640625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0049729375168681145,
"kl": 0.00047132828331086785,
"learning_rate": 2.564102564102564e-06,
"loss": 4.8525371312280186e-06,
"num_tokens": 98852.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 21,
"step_time": 6.948417052999957
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 24.75,
"completions/mean_terminated_length": 24.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09643011912703514,
"epoch": 0.171875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006859563756734133,
"kl": 0.0007727623451501131,
"learning_rate": 2.6923076923076923e-06,
"loss": 6.733347163390135e-06,
"num_tokens": 103734.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 22,
"step_time": 7.359714186000019
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 47.0,
"completions/max_terminated_length": 47.0,
"completions/mean_length": 23.375,
"completions/mean_terminated_length": 23.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.42096097022295,
"epoch": 0.1796875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005077087786048651,
"kl": 0.0007580214296467602,
"learning_rate": 2.8205128205128207e-06,
"loss": 7.698860827076714e-06,
"num_tokens": 108625.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 23,
"step_time": 8.31426027000009
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 30.0,
"completions/mean_terminated_length": 30.0,
"completions/min_length": 30.0,
"completions/min_terminated_length": 30.0,
"entropy": 0.05439547263085842,
"epoch": 0.1875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.000995291629806161,
"kl": 0.0004853812279179692,
"learning_rate": 2.948717948717949e-06,
"loss": 4.853812242799904e-06,
"num_tokens": 112781.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 24,
"step_time": 5.9903554890000805
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 37.0,
"completions/max_terminated_length": 37.0,
"completions/mean_length": 24.375,
"completions/mean_terminated_length": 24.375,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"entropy": 0.10691225156188011,
"epoch": 0.1953125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.011116056703031063,
"kl": 0.0012052947713527828,
"learning_rate": 3.0769230769230774e-06,
"loss": 1.1946971426368691e-05,
"num_tokens": 117612.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 25,
"step_time": 7.288805328999956
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 25.0,
"completions/mean_terminated_length": 25.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1299378089606762,
"epoch": 0.203125,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.851197361946106,
"kl": 0.002050661132670939,
"learning_rate": 3.205128205128206e-06,
"loss": -0.11245696991682053,
"num_tokens": 122448.0,
"reward": 0.875,
"reward_std": 0.3535533845424652,
"rewards/reward_fn/mean": 0.875,
"rewards/reward_fn/std": 0.3535533845424652,
"step": 26,
"step_time": 7.037999247000016
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.375,
"completions/mean_terminated_length": 21.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07765422388911247,
"epoch": 0.2109375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00900979246944189,
"kl": 0.001701147179119289,
"learning_rate": 3.3333333333333333e-06,
"loss": 1.7008303984766826e-05,
"num_tokens": 127211.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 27,
"step_time": 6.768718518000014
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 28.0,
"completions/mean_terminated_length": 28.0,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"entropy": 0.053295012563467026,
"epoch": 0.21875,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.0311310291290283,
"kl": 0.005389484780607745,
"learning_rate": 3.4615384615384617e-06,
"loss": -0.10706700384616852,
"num_tokens": 131351.0,
"reward": 0.893750011920929,
"reward_std": 0.3005203604698181,
"rewards/reward_fn/mean": 0.893750011920929,
"rewards/reward_fn/std": 0.3005203604698181,
"step": 28,
"step_time": 5.859235952000063
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 18.5,
"completions/mean_terminated_length": 18.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08729634806513786,
"epoch": 0.2265625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.11862029135227203,
"kl": 0.00995962810702622,
"learning_rate": 3.58974358974359e-06,
"loss": 0.00010122068488271907,
"num_tokens": 136067.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 29,
"step_time": 7.392926845999909
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 25.125,
"completions/mean_terminated_length": 25.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.049397675320506096,
"epoch": 0.234375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.013916196301579475,
"kl": 0.004311037337174639,
"learning_rate": 3.7179487179487184e-06,
"loss": 3.796327655436471e-05,
"num_tokens": 140216.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 30,
"step_time": 5.707905451999977
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 128.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 44.625,
"completions/mean_terminated_length": 32.71428680419922,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"entropy": 0.3870660662651062,
"epoch": 0.2421875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00857287272810936,
"kl": 0.0027850136975757778,
"learning_rate": 3.846153846153847e-06,
"loss": 2.3635355319129303e-05,
"num_tokens": 145949.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 31,
"step_time": 14.826277505999997
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 29.25,
"completions/mean_terminated_length": 29.25,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"entropy": 0.15847552567720413,
"epoch": 0.25,
"frac_reward_zero_std": 0.5,
"grad_norm": 3.0680534839630127,
"kl": 0.005178321152925491,
"learning_rate": 3.974358974358974e-06,
"loss": 0.021413102746009827,
"num_tokens": 150875.0,
"reward": 0.875,
"reward_std": 0.3535533845424652,
"rewards/reward_fn/mean": 0.875,
"rewards/reward_fn/std": 0.3535533845424652,
"step": 32,
"step_time": 7.193657256999927
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.375,
"completions/mean_terminated_length": 27.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.036486370489001274,
"epoch": 0.2578125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.013429846614599228,
"kl": 0.0038138862000778317,
"learning_rate": 4.102564102564103e-06,
"loss": 3.6106022889725864e-05,
"num_tokens": 155090.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 33,
"step_time": 5.989709379999908
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 24.75,
"completions/mean_terminated_length": 24.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06528343632817268,
"epoch": 0.265625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.23420189321041107,
"kl": 0.015011879149824381,
"learning_rate": 4.230769230769231e-06,
"loss": 0.00015618561883457005,
"num_tokens": 159860.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 34,
"step_time": 7.343351643999995
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 24.75,
"completions/mean_terminated_length": 24.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09101804345846176,
"epoch": 0.2734375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0637916773557663,
"kl": 0.011130816768854856,
"learning_rate": 4.358974358974359e-06,
"loss": 0.00011102524877060205,
"num_tokens": 165430.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 35,
"step_time": 7.710919977999993
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 27.0,
"completions/mean_terminated_length": 27.0,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"entropy": 0.06510375998914242,
"epoch": 0.28125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03826780989766121,
"kl": 0.0053558857180178165,
"learning_rate": 4.487179487179488e-06,
"loss": 5.355885878088884e-05,
"num_tokens": 170254.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 36,
"step_time": 7.2804435799999965
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 27.125,
"completions/mean_terminated_length": 27.125,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"entropy": 0.043790558353066444,
"epoch": 0.2890625,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.402679681777954,
"kl": 0.02490220731124282,
"learning_rate": 4.615384615384616e-06,
"loss": 0.03481510281562805,
"num_tokens": 174395.0,
"reward": 0.875,
"reward_std": 0.3535533845424652,
"rewards/reward_fn/mean": 0.875,
"rewards/reward_fn/std": 0.3535533845424652,
"step": 37,
"step_time": 5.756117628000084
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 18.875,
"completions/mean_terminated_length": 18.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12046026065945625,
"epoch": 0.296875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.054251085966825485,
"kl": 0.018674671184271574,
"learning_rate": 4.743589743589744e-06,
"loss": 0.00016035939916037023,
"num_tokens": 179942.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 38,
"step_time": 7.3252331490000415
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 26.375,
"completions/mean_terminated_length": 26.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05612089857459068,
"epoch": 0.3046875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.039188120514154434,
"kl": 0.012932237819768488,
"learning_rate": 4.871794871794872e-06,
"loss": 0.00010084287350764498,
"num_tokens": 184741.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 39,
"step_time": 7.515498236999974
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 24.75,
"completions/mean_terminated_length": 24.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09634075127542019,
"epoch": 0.3125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.018983978778123856,
"kl": 0.005221477011218667,
"learning_rate": 5e-06,
"loss": 5.3893018048256636e-05,
"num_tokens": 189627.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 40,
"step_time": 7.176626092999982
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.25,
"completions/mean_terminated_length": 21.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06109755299985409,
"epoch": 0.3203125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.030976427718997,
"kl": 0.009227571543306112,
"learning_rate": 4.999896350176413e-06,
"loss": 9.208174014929682e-05,
"num_tokens": 194497.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 41,
"step_time": 6.542497393999952
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 20.625,
"completions/mean_terminated_length": 20.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.17987589165568352,
"epoch": 0.328125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0431177020072937,
"kl": 0.015479911118745804,
"learning_rate": 4.999585409300281e-06,
"loss": 0.0001403249625582248,
"num_tokens": 200082.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 42,
"step_time": 7.701265004999868
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 25.125,
"completions/mean_terminated_length": 25.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07510707713663578,
"epoch": 0.3359375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03998936340212822,
"kl": 0.012762806843966246,
"learning_rate": 4.999067203154777e-06,
"loss": 0.00010049781849374995,
"num_tokens": 204927.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 43,
"step_time": 7.225048154999968
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.5,
"completions/mean_terminated_length": 19.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06112176366150379,
"epoch": 0.34375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.028579436242580414,
"kl": 0.01107706013135612,
"learning_rate": 4.998341774709482e-06,
"loss": 0.00010365620255470276,
"num_tokens": 209771.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 44,
"step_time": 6.643636920000063
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 44.0,
"completions/max_terminated_length": 44.0,
"completions/mean_length": 27.875,
"completions/mean_terminated_length": 27.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12446310371160507,
"epoch": 0.3515625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.012319399043917656,
"kl": 0.004381780279800296,
"learning_rate": 4.9974091841168195e-06,
"loss": 4.3796011595986784e-05,
"num_tokens": 214682.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 45,
"step_time": 7.909422166999889
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 22.375,
"completions/mean_terminated_length": 22.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.10363675281405449,
"epoch": 0.359375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.04150591790676117,
"kl": 0.007062526885420084,
"learning_rate": 4.99626950870707e-06,
"loss": 6.894840043969452e-05,
"num_tokens": 220237.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 46,
"step_time": 7.691492860000039
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.5,
"completions/mean_terminated_length": 27.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.030814163386821747,
"epoch": 0.3671875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.013106024824082851,
"kl": 0.0039410877507179976,
"learning_rate": 4.994922842981958e-06,
"loss": 3.779985854635015e-05,
"num_tokens": 224417.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 47,
"step_time": 5.999039098000026
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 16.625,
"completions/mean_terminated_length": 16.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1171766109764576,
"epoch": 0.375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.04344555363059044,
"kl": 0.010649031959474087,
"learning_rate": 4.993369298606817e-06,
"loss": 0.00010028588440036401,
"num_tokens": 229926.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 48,
"step_time": 7.581720861000008
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.625,
"completions/mean_terminated_length": 13.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1510508954524994,
"epoch": 0.3828125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.04870390519499779,
"kl": 0.011529957875609398,
"learning_rate": 4.991609004401324e-06,
"loss": 0.0001149907911894843,
"num_tokens": 235435.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 49,
"step_time": 5.576136509999969
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.0,
"completions/mean_terminated_length": 17.0,
"completions/min_length": 12.0,
"completions/min_terminated_length": 12.0,
"entropy": 0.13440731912851334,
"epoch": 0.390625,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.8115813732147217,
"kl": 0.20483593340031803,
"learning_rate": 4.989642106328829e-06,
"loss": -0.1266314685344696,
"num_tokens": 240183.0,
"reward": 0.875,
"reward_std": 0.3535533845424652,
"rewards/reward_fn/mean": 0.875,
"rewards/reward_fn/std": 0.3535533845424652,
"step": 50,
"step_time": 6.442843831000005
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.375,
"completions/mean_terminated_length": 27.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.029575850814580917,
"epoch": 0.3984375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006918889936059713,
"kl": 0.0022675381042063236,
"learning_rate": 4.98746876748424e-06,
"loss": 2.1996882423991337e-05,
"num_tokens": 244150.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 51,
"step_time": 5.648842897999998
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.75,
"completions/mean_terminated_length": 13.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12618596106767654,
"epoch": 0.40625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.04183642938733101,
"kl": 0.007541614351794124,
"learning_rate": 4.985089168080509e-06,
"loss": 7.499127241317183e-05,
"num_tokens": 249636.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 52,
"step_time": 5.985882786999923
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 27.0,
"completions/mean_terminated_length": 27.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.027834360487759113,
"epoch": 0.4140625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005657592322677374,
"kl": 0.0021601368789561093,
"learning_rate": 4.982503505433683e-06,
"loss": 2.106852480210364e-05,
"num_tokens": 253604.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 53,
"step_time": 5.815157560000102
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 49.0,
"completions/max_terminated_length": 49.0,
"completions/mean_length": 30.0,
"completions/mean_terminated_length": 30.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.049247266724705696,
"epoch": 0.421875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008218275383114815,
"kl": 0.0025316422106698155,
"learning_rate": 4.979711993946543e-06,
"loss": 2.4405037038377486e-05,
"num_tokens": 257780.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 54,
"step_time": 7.388005351000061
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 128.0,
"completions/max_terminated_length": 71.0,
"completions/mean_length": 38.25,
"completions/mean_terminated_length": 25.428571701049805,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"entropy": 0.14161667972803116,
"epoch": 0.4296875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.011416819877922535,
"kl": 0.0019309421186335385,
"learning_rate": 4.976714865090827e-06,
"loss": 2.0552859496092424e-05,
"num_tokens": 263510.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 55,
"step_time": 15.045006353999952
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 19.125,
"completions/mean_terminated_length": 19.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11700041592121124,
"epoch": 0.4375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01164371706545353,
"kl": 0.001374046492855996,
"learning_rate": 4.973512367388038e-06,
"loss": 1.2746955690090545e-05,
"num_tokens": 269087.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 56,
"step_time": 7.920732168999962
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 94.0,
"completions/max_terminated_length": 94.0,
"completions/mean_length": 37.625,
"completions/mean_terminated_length": 37.625,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"entropy": 0.04022482968866825,
"epoch": 0.4453125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0068351044319570065,
"kl": 0.0017894834163598716,
"learning_rate": 4.970104766388833e-06,
"loss": 1.8479673599358648e-05,
"num_tokens": 273396.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 57,
"step_time": 11.183577817000014
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 23.25,
"completions/mean_terminated_length": 23.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06758509948849678,
"epoch": 0.453125,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.572488784790039,
"kl": 0.0030532198143191636,
"learning_rate": 4.966492344651006e-06,
"loss": -0.07791005074977875,
"num_tokens": 278234.0,
"reward": 0.8812500238418579,
"reward_std": 0.3358757197856903,
"rewards/reward_fn/mean": 0.8812500238418579,
"rewards/reward_fn/std": 0.3358757197856903,
"step": 58,
"step_time": 7.65875285900006
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 27.0,
"completions/mean_terminated_length": 27.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.033227477222681046,
"epoch": 0.4609375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0016361080342903733,
"kl": 0.0016750092036090791,
"learning_rate": 4.962675401716056e-06,
"loss": 1.6619302186882123e-05,
"num_tokens": 282454.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 59,
"step_time": 5.975613750999969
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.375,
"completions/mean_terminated_length": 13.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13986211270093918,
"epoch": 0.46875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.04219949617981911,
"kl": 0.004549495875835419,
"learning_rate": 4.958654254084356e-06,
"loss": 4.563133552437648e-05,
"num_tokens": 287961.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 60,
"step_time": 5.713278319999972
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.25,
"completions/mean_terminated_length": 21.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0659055095165968,
"epoch": 0.4765625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.010506045073270798,
"kl": 0.0030614889692515135,
"learning_rate": 4.954429235188897e-06,
"loss": 2.8885815481771715e-05,
"num_tokens": 292807.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 61,
"step_time": 6.593735059999972
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 25.375,
"completions/mean_terminated_length": 25.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.040277909487485886,
"epoch": 0.484375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006953485310077667,
"kl": 0.0017460978706367314,
"learning_rate": 4.95000069536765e-06,
"loss": 1.6949392374954186e-05,
"num_tokens": 296982.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 62,
"step_time": 5.818303646000004
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.375,
"completions/mean_terminated_length": 13.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.14363406598567963,
"epoch": 0.4921875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0677562728524208,
"kl": 0.00942042376846075,
"learning_rate": 4.9453690018345144e-06,
"loss": 9.429677447769791e-05,
"num_tokens": 302489.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 63,
"step_time": 5.593983074999983
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.25,
"completions/mean_terminated_length": 13.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12100231274962425,
"epoch": 0.5,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01809072121977806,
"kl": 0.003618443734012544,
"learning_rate": 4.940534538648862e-06,
"loss": 3.6511512007564306e-05,
"num_tokens": 308019.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 64,
"step_time": 5.77941624999994
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.5,
"completions/mean_terminated_length": 27.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03561065532267094,
"epoch": 0.5078125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0022499635815620422,
"kl": 0.0012880651047453284,
"learning_rate": 4.935497706683698e-06,
"loss": 1.28087995108217e-05,
"num_tokens": 312083.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 65,
"step_time": 5.752014847000055
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 15.25,
"completions/mean_terminated_length": 15.25,
"completions/min_length": 12.0,
"completions/min_terminated_length": 12.0,
"entropy": 0.14075982943177223,
"epoch": 0.515625,
"frac_reward_zero_std": 0.5,
"grad_norm": 2.6456313133239746,
"kl": 0.09695499250665307,
"learning_rate": 4.9302589235924185e-06,
"loss": -0.08113337308168411,
"num_tokens": 316841.0,
"reward": 0.875,
"reward_std": 0.3535533845424652,
"rewards/reward_fn/mean": 0.875,
"rewards/reward_fn/std": 0.3535533845424652,
"step": 66,
"step_time": 6.563686877999999
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.125,
"completions/mean_terminated_length": 19.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07007651403546333,
"epoch": 0.5234375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.017309362068772316,
"kl": 0.004948096117004752,
"learning_rate": 4.924818623774178e-06,
"loss": 4.838653694605455e-05,
"num_tokens": 321630.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 67,
"step_time": 6.570541892999927
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 23.375,
"completions/mean_terminated_length": 23.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0408208854496479,
"epoch": 0.53125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.012536097317934036,
"kl": 0.004252667655237019,
"learning_rate": 4.91917725833787e-06,
"loss": 3.6156445275992155e-05,
"num_tokens": 325569.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 68,
"step_time": 5.749754669000026
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 13.0,
"completions/max_terminated_length": 13.0,
"completions/mean_length": 13.0,
"completions/mean_terminated_length": 13.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1359182596206665,
"epoch": 0.5390625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03752928227186203,
"kl": 0.009188917931169271,
"learning_rate": 4.913335295064721e-06,
"loss": 9.188917465507984e-05,
"num_tokens": 331045.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 69,
"step_time": 5.468208492999906
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.5,
"completions/mean_terminated_length": 19.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07358374819159508,
"epoch": 0.546875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.020383596420288086,
"kl": 0.005584244150668383,
"learning_rate": 4.907293218369499e-06,
"loss": 5.529334521270357e-05,
"num_tokens": 335761.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 70,
"step_time": 6.64621212499992
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.0,
"completions/mean_terminated_length": 17.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05005803145468235,
"epoch": 0.5546875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03125562518835068,
"kl": 0.008841587696224451,
"learning_rate": 4.901051529260352e-06,
"loss": 8.841587987262756e-05,
"num_tokens": 339757.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 71,
"step_time": 5.899700159999952
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 17.25,
"completions/mean_terminated_length": 17.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07237722724676132,
"epoch": 0.5625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.018481554463505745,
"kl": 0.006117145298048854,
"learning_rate": 4.89461074529726e-06,
"loss": 6.117144948802888e-05,
"num_tokens": 344555.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 72,
"step_time": 6.860756025000001
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.0,
"completions/mean_terminated_length": 21.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.045137686654925346,
"epoch": 0.5703125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.020073302090168,
"kl": 0.005907518556341529,
"learning_rate": 4.8879714005491205e-06,
"loss": 5.907518061576411e-05,
"num_tokens": 348591.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 73,
"step_time": 5.62259969999991
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.125,
"completions/mean_terminated_length": 19.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0720517449080944,
"epoch": 0.578125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.027030251920223236,
"kl": 0.01036432757973671,
"learning_rate": 4.881134045549463e-06,
"loss": 8.4122279076837e-05,
"num_tokens": 353312.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 74,
"step_time": 6.612577034999958
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 128.0,
"completions/max_terminated_length": 13.0,
"completions/mean_length": 27.375,
"completions/mean_terminated_length": 13.000000953674316,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1056247390806675,
"epoch": 0.5859375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.019467800855636597,
"kl": 0.006749335676431656,
"learning_rate": 4.874099247250799e-06,
"loss": 5.878092997591011e-05,
"num_tokens": 358931.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 75,
"step_time": 14.855594667999867
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 17.125,
"completions/mean_terminated_length": 17.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.048843516036868095,
"epoch": 0.59375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.033446282148361206,
"kl": 0.012164841871708632,
"learning_rate": 4.8668675889776095e-06,
"loss": 0.00012165858061052859,
"num_tokens": 362844.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 76,
"step_time": 5.699464158999945
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.0,
"completions/mean_terminated_length": 17.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07759269699454308,
"epoch": 0.6015625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.017857957631349564,
"kl": 0.0074398701544851065,
"learning_rate": 4.85943967037798e-06,
"loss": 6.409453635569662e-05,
"num_tokens": 367560.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 77,
"step_time": 6.886273725999672
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 15.875,
"completions/mean_terminated_length": 15.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12127615511417389,
"epoch": 0.609375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.024566544219851494,
"kl": 0.007913533365353942,
"learning_rate": 4.851816107373871e-06,
"loss": 7.854383147787303e-05,
"num_tokens": 373083.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 78,
"step_time": 7.353863909999973
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.25,
"completions/mean_terminated_length": 13.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13099069893360138,
"epoch": 0.6171875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0477265790104866,
"kl": 0.011370216961950064,
"learning_rate": 4.843997532110051e-06,
"loss": 0.00011370217544026673,
"num_tokens": 378589.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 79,
"step_time": 5.622442682000155
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07205060124397278,
"epoch": 0.625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.012449050322175026,
"kl": 0.00409984530415386,
"learning_rate": 4.835984592901678e-06,
"loss": 3.9597347495146096e-05,
"num_tokens": 383357.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 80,
"step_time": 6.542991357999881
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 21.875,
"completions/mean_terminated_length": 21.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07283683493733406,
"epoch": 0.6328125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.019634408876299858,
"kl": 0.007241100538522005,
"learning_rate": 4.82777795418054e-06,
"loss": 7.439294131472707e-05,
"num_tokens": 388152.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 81,
"step_time": 7.132434741999987
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 15.25,
"completions/mean_terminated_length": 15.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08575013279914856,
"epoch": 0.640625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03768326714634895,
"kl": 0.010142359882593155,
"learning_rate": 4.819378296439962e-06,
"loss": 0.00010288630437571555,
"num_tokens": 392866.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 82,
"step_time": 6.954328571000133
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 13.0,
"completions/max_terminated_length": 13.0,
"completions/mean_length": 13.0,
"completions/mean_terminated_length": 13.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12997420877218246,
"epoch": 0.6484375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01888282783329487,
"kl": 0.005712541053071618,
"learning_rate": 4.810786316178377e-06,
"loss": 5.712541315006092e-05,
"num_tokens": 398394.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 83,
"step_time": 6.931600935000006
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 22.125,
"completions/mean_terminated_length": 22.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06609366461634636,
"epoch": 0.65625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.017994992434978485,
"kl": 0.00510314735583961,
"learning_rate": 4.802002725841577e-06,
"loss": 4.8029709432739764e-05,
"num_tokens": 403295.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 84,
"step_time": 7.442147588000125
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 42.0,
"completions/max_terminated_length": 42.0,
"completions/mean_length": 18.625,
"completions/mean_terminated_length": 18.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11900537833571434,
"epoch": 0.6640625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.024866754189133644,
"kl": 0.006785314995795488,
"learning_rate": 4.793028253763633e-06,
"loss": 7.137990178307518e-05,
"num_tokens": 408052.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 85,
"step_time": 7.643961140000329
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 16.125,
"completions/mean_terminated_length": 16.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.10917261242866516,
"epoch": 0.671875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.022167593240737915,
"kl": 0.0044796590227633715,
"learning_rate": 4.783863644106502e-06,
"loss": 4.586868089972995e-05,
"num_tokens": 413581.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 86,
"step_time": 7.9875519340000665
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06199129670858383,
"epoch": 0.6796875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03173369914293289,
"kl": 0.008890938945114613,
"learning_rate": 4.774509656798326e-06,
"loss": 8.7326108769048e-05,
"num_tokens": 418435.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 87,
"step_time": 6.503118833999906
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 25.375,
"completions/mean_terminated_length": 25.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0360421109944582,
"epoch": 0.6875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01023577805608511,
"kl": 0.003528562723658979,
"learning_rate": 4.764967067470409e-06,
"loss": 3.279188240412623e-05,
"num_tokens": 422510.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 88,
"step_time": 5.7346701770002255
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.25,
"completions/mean_terminated_length": 13.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13792025297880173,
"epoch": 0.6953125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.036804474890232086,
"kl": 0.007783198030665517,
"learning_rate": 4.755236667392914e-06,
"loss": 7.783197361277416e-05,
"num_tokens": 428016.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 89,
"step_time": 5.881843299000138
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 29.0,
"completions/mean_terminated_length": 29.0,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"entropy": 0.033070383593440056,
"epoch": 0.703125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.002256783191114664,
"kl": 0.001209279813338071,
"learning_rate": 4.745319263409241e-06,
"loss": 1.2092797987861559e-05,
"num_tokens": 432116.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 90,
"step_time": 5.656249395999794
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 128.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 36.25,
"completions/mean_terminated_length": 23.142858505249023,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.2254822440445423,
"epoch": 0.7109375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.014093738049268723,
"kl": 0.0030488879419863224,
"learning_rate": 4.735215677869129e-06,
"loss": 3.228533023502678e-05,
"num_tokens": 437806.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 91,
"step_time": 14.648245948000067
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 26.75,
"completions/mean_terminated_length": 26.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05311024375259876,
"epoch": 0.71875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00612290482968092,
"kl": 0.0022808221401646733,
"learning_rate": 4.724926748560464e-06,
"loss": 2.2808220819570124e-05,
"num_tokens": 442648.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 92,
"step_time": 7.179930319999812
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 19.375,
"completions/mean_terminated_length": 19.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11795984953641891,
"epoch": 0.7265625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.014634083956480026,
"kl": 0.0032170950435101986,
"learning_rate": 4.714453328639814e-06,
"loss": 3.587050741771236e-05,
"num_tokens": 448179.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 93,
"step_time": 7.483695861999877
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 19.875,
"completions/mean_terminated_length": 19.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07148051261901855,
"epoch": 0.734375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.010934879072010517,
"kl": 0.0029092643526382744,
"learning_rate": 4.7037962865616795e-06,
"loss": 2.7483838493935764e-05,
"num_tokens": 453030.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 94,
"step_time": 7.076518674999988
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 22.125,
"completions/mean_terminated_length": 22.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06901054829359055,
"epoch": 0.7421875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006595511920750141,
"kl": 0.0014058846863918006,
"learning_rate": 4.692956506006486e-06,
"loss": 1.403903206664836e-05,
"num_tokens": 457855.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 95,
"step_time": 7.476475853000011
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.125,
"completions/mean_terminated_length": 17.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08090204373002052,
"epoch": 0.75,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.020713720470666885,
"kl": 0.005518489051610231,
"learning_rate": 4.681934885807307e-06,
"loss": 5.53158279217314e-05,
"num_tokens": 462588.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 96,
"step_time": 6.740178639000078
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 23.0,
"completions/mean_terminated_length": 23.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.037110449746251106,
"epoch": 0.7578125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006115948315709829,
"kl": 0.0022233356721699238,
"learning_rate": 4.6707323398753346e-06,
"loss": 2.206304998253472e-05,
"num_tokens": 466604.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 97,
"step_time": 5.532072194999955
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06933313608169556,
"epoch": 0.765625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007001116406172514,
"kl": 0.0019759326823987067,
"learning_rate": 4.659349797124096e-06,
"loss": 1.9955001334892586e-05,
"num_tokens": 471382.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 98,
"step_time": 6.4825494700000945
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 25.5,
"completions/mean_terminated_length": 25.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03348589688539505,
"epoch": 0.7734375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.002566123381257057,
"kl": 0.0018573449924588203,
"learning_rate": 4.647788201392429e-06,
"loss": 1.857344977906905e-05,
"num_tokens": 475554.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 99,
"step_time": 5.890160061000188
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.25,
"completions/max_length": 128.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 42.0,
"completions/mean_terminated_length": 13.333333969116211,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.21250910684466362,
"epoch": 0.78125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.07006637006998062,
"kl": 0.017085027415305376,
"learning_rate": 4.636048511366222e-06,
"loss": 0.00017085025319829583,
"num_tokens": 481290.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 100,
"step_time": 14.571886820999907
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 16.25,
"completions/mean_terminated_length": 16.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11931024491786957,
"epoch": 0.7890625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0073622651398181915,
"kl": 0.0012832598586101085,
"learning_rate": 4.624131700498913e-06,
"loss": 1.3542096894525457e-05,
"num_tokens": 486820.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 101,
"step_time": 7.506882940999958
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 128.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 33.75,
"completions/mean_terminated_length": 20.285715103149414,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05769490636885166,
"epoch": 0.796875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.002226189011707902,
"kl": 0.001102528884075582,
"learning_rate": 4.612038756930778e-06,
"loss": 8.816463378025219e-06,
"num_tokens": 491814.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 102,
"step_time": 14.50194374800003
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.375,
"completions/mean_terminated_length": 27.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.032190028578042984,
"epoch": 0.8046875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0019165624398738146,
"kl": 0.0013525813119485974,
"learning_rate": 4.599770683406992e-06,
"loss": 1.3399311683315318e-05,
"num_tokens": 495785.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 103,
"step_time": 5.6954472240001905
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 22.125,
"completions/mean_terminated_length": 22.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09382087737321854,
"epoch": 0.8125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0029365946538746357,
"kl": 0.002741978387348354,
"learning_rate": 4.587328497194478e-06,
"loss": 2.726353341131471e-05,
"num_tokens": 501338.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 104,
"step_time": 7.473332564999964
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 26.75,
"completions/mean_terminated_length": 26.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05792428180575371,
"epoch": 0.8203125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0022111833095550537,
"kl": 0.0035839949268847704,
"learning_rate": 4.5747132299975634e-06,
"loss": 4.1183855501003563e-05,
"num_tokens": 506124.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 105,
"step_time": 7.173797810999986
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.625,
"completions/mean_terminated_length": 21.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06740124337375164,
"epoch": 0.828125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0050131166353821754,
"kl": 0.0017226869240403175,
"learning_rate": 4.561925927872421e-06,
"loss": 1.599166716914624e-05,
"num_tokens": 510953.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 106,
"step_time": 6.916509818999884
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.125,
"completions/mean_terminated_length": 21.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07052117213606834,
"epoch": 0.8359375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007340835873037577,
"kl": 0.0017220252193510532,
"learning_rate": 4.548967651140341e-06,
"loss": 1.7247453797608614e-05,
"num_tokens": 515806.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 107,
"step_time": 6.615679707000027
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 24.125,
"completions/mean_terminated_length": 24.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06415174063295126,
"epoch": 0.84375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0005465157446451485,
"kl": 0.001015377274597995,
"learning_rate": 4.5358394742998e-06,
"loss": 1.1552911928447429e-05,
"num_tokens": 520699.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 108,
"step_time": 7.493365885000003
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.25,
"completions/mean_terminated_length": 21.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0641067698597908,
"epoch": 0.8515625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.010530305095016956,
"kl": 0.0026437516789883375,
"learning_rate": 4.522542485937369e-06,
"loss": 2.6550886104814708e-05,
"num_tokens": 525437.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 109,
"step_time": 6.527635427999712
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 25.75,
"completions/mean_terminated_length": 25.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0577393751591444,
"epoch": 0.859375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008794880472123623,
"kl": 0.0028055842267349362,
"learning_rate": 4.509077788637446e-06,
"loss": 2.7364983907318674e-05,
"num_tokens": 530267.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 110,
"step_time": 7.3437313919998815
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 24.0,
"completions/mean_terminated_length": 24.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0586474034935236,
"epoch": 0.8671875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005623109173029661,
"kl": 0.0036008708993904293,
"learning_rate": 4.4954464988908306e-06,
"loss": 3.5230626963311806e-05,
"num_tokens": 535131.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 111,
"step_time": 7.285259039999801
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 16.375,
"completions/mean_terminated_length": 16.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12351097911596298,
"epoch": 0.875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01802302524447441,
"kl": 0.0029585817828774452,
"learning_rate": 4.481649747002146e-06,
"loss": 3.03143824567087e-05,
"num_tokens": 540638.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 112,
"step_time": 7.425657803999911
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.625,
"completions/mean_terminated_length": 19.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06710582785308361,
"epoch": 0.8828125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0028362085577100515,
"kl": 0.0015989196253940463,
"learning_rate": 4.467688676996111e-06,
"loss": 1.6505855455761775e-05,
"num_tokens": 545519.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 113,
"step_time": 6.964208683000152
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.5,
"completions/mean_terminated_length": 19.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07923074252903461,
"epoch": 0.890625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008551168255507946,
"kl": 0.002287313051056117,
"learning_rate": 4.4535644465226795e-06,
"loss": 1.9634611817309633e-05,
"num_tokens": 550319.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 114,
"step_time": 6.80012675099988
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 20.0,
"completions/mean_terminated_length": 20.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07435985282063484,
"epoch": 0.8984375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.014601176604628563,
"kl": 0.005440297303721309,
"learning_rate": 4.43927822676105e-06,
"loss": 4.615134821506217e-05,
"num_tokens": 555091.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 115,
"step_time": 7.212101119999943
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08148583956062794,
"epoch": 0.90625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004445843398571014,
"kl": 0.001686684088781476,
"learning_rate": 4.424831202322548e-06,
"loss": 1.5020732462289743e-05,
"num_tokens": 559873.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 116,
"step_time": 6.573965233000081
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 30.875,
"completions/mean_terminated_length": 30.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05395838990807533,
"epoch": 0.9140625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.002361060120165348,
"kl": 0.0016205881838686764,
"learning_rate": 4.410224571152402e-06,
"loss": 1.6156718629645184e-05,
"num_tokens": 564700.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 117,
"step_time": 7.449374204000151
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 25.5,
"completions/mean_terminated_length": 25.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03191390447318554,
"epoch": 0.921875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.04630071669816971,
"kl": 0.015495602798182517,
"learning_rate": 4.395459544430407e-06,
"loss": 0.00015495602565351874,
"num_tokens": 568832.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 118,
"step_time": 5.798940031999791
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.375,
"completions/mean_terminated_length": 19.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08851146325469017,
"epoch": 0.9296875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03349859267473221,
"kl": 0.0069463985855691135,
"learning_rate": 4.380537346470495e-06,
"loss": 7.56498338887468e-05,
"num_tokens": 573559.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 119,
"step_time": 6.745357660999844
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.041911693289875984,
"epoch": 0.9375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.003103650640696287,
"kl": 0.0021437766263261437,
"learning_rate": 4.3654592146192146e-06,
"loss": 2.1203726646490395e-05,
"num_tokens": 577735.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 120,
"step_time": 5.838892162999855
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 24.5,
"completions/mean_terminated_length": 24.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0581835713237524,
"epoch": 0.9453125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0017389818094670773,
"kl": 0.00121387725812383,
"learning_rate": 4.35022639915313e-06,
"loss": 1.2284486729186028e-05,
"num_tokens": 582627.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 121,
"step_time": 7.239668368000139
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 29.0,
"completions/mean_terminated_length": 29.0,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"entropy": 0.027326339855790138,
"epoch": 0.953125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00015551889373455197,
"kl": 0.0014788485132157803,
"learning_rate": 4.334840163175152e-06,
"loss": 1.4788485714234412e-05,
"num_tokens": 586899.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 122,
"step_time": 5.908869453000079
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.125,
"completions/mean_terminated_length": 19.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08600620739161968,
"epoch": 0.9609375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004444346763193607,
"kl": 0.0013841381878592074,
"learning_rate": 4.319301782509794e-06,
"loss": 1.4433127944357693e-05,
"num_tokens": 591680.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 123,
"step_time": 6.542298487999915
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 29.75,
"completions/mean_terminated_length": 29.75,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"entropy": 0.05058911070227623,
"epoch": 0.96875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.002049674978479743,
"kl": 0.0011390995350666344,
"learning_rate": 4.30361254559739e-06,
"loss": 1.1433874533395283e-05,
"num_tokens": 596474.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 124,
"step_time": 7.415240068999992
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 20.25,
"completions/mean_terminated_length": 20.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.14710266143083572,
"epoch": 0.9765625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02711130492389202,
"kl": 0.0030616128933615983,
"learning_rate": 4.287773753387249e-06,
"loss": 3.7990492273820564e-05,
"num_tokens": 602012.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 125,
"step_time": 7.445253116999993
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 40.0,
"completions/max_terminated_length": 40.0,
"completions/mean_length": 25.625,
"completions/mean_terminated_length": 25.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.060312340036034584,
"epoch": 0.984375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004181354772299528,
"kl": 0.0010917744366452098,
"learning_rate": 4.271786719229787e-06,
"loss": 1.0562279385339934e-05,
"num_tokens": 606797.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 126,
"step_time": 7.632959243999949
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 128.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 27.75,
"completions/mean_terminated_length": 13.428571701049805,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.10784962400794029,
"epoch": 0.9921875,
"frac_reward_zero_std": 1.0,
"grad_norm": 1.5359536409378052,
"kl": 0.055240524452528916,
"learning_rate": 4.255652768767619e-06,
"loss": 0.0008335658931173384,
"num_tokens": 612419.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 127,
"step_time": 14.956502908999937
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 57.0,
"completions/max_terminated_length": 57.0,
"completions/mean_length": 32.125,
"completions/mean_terminated_length": 32.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09433136507868767,
"epoch": 1.0,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.026125041767954826,
"kl": 0.005883891310077161,
"learning_rate": 4.23937323982564e-06,
"loss": 6.240967923076823e-05,
"num_tokens": 617328.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 128,
"step_time": 8.913310376000027
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 25.0,
"completions/mean_terminated_length": 25.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03300720080733299,
"epoch": 1.0078125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0028305482119321823,
"kl": 0.001693042810074985,
"learning_rate": 4.222949482300094e-06,
"loss": 1.6930427591432817e-05,
"num_tokens": 621356.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 129,
"step_time": 5.800293608999937
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07437415793538094,
"epoch": 1.015625,
"frac_reward_zero_std": 0.5,
"grad_norm": 3.94826078414917,
"kl": 2.5161777958273888,
"learning_rate": 4.206382858046636e-06,
"loss": -0.11797107756137848,
"num_tokens": 626170.0,
"reward": 0.887499988079071,
"reward_std": 0.3181980550289154,
"rewards/reward_fn/mean": 0.887499988079071,
"rewards/reward_fn/std": 0.3181980550289154,
"step": 130,
"step_time": 6.945177734999788
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 17.375,
"completions/mean_terminated_length": 17.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07938191294670105,
"epoch": 1.0234375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00952488835901022,
"kl": 0.001886414596810937,
"learning_rate": 4.189674740767411e-06,
"loss": 1.6817120922496542e-05,
"num_tokens": 630937.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 131,
"step_time": 6.6918233229998805
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.25,
"completions/mean_terminated_length": 17.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07850120961666107,
"epoch": 1.03125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.013484103605151176,
"kl": 0.0045427161967381835,
"learning_rate": 4.172826515897146e-06,
"loss": 4.5427161239786074e-05,
"num_tokens": 635691.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 132,
"step_time": 6.658873882999842
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08399690128862858,
"epoch": 1.0390625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.025045400485396385,
"kl": 0.029486034996807575,
"learning_rate": 4.15583958048827e-06,
"loss": 0.00029239041032269597,
"num_tokens": 640561.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 133,
"step_time": 6.894192197999928
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.125,
"completions/mean_terminated_length": 13.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13748392462730408,
"epoch": 1.046875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01748264580965042,
"kl": 0.001855719368904829,
"learning_rate": 4.138715343095069e-06,
"loss": 1.858307223301381e-05,
"num_tokens": 646038.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 134,
"step_time": 5.438569751999921
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 23.125,
"completions/mean_terminated_length": 23.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.04659535735845566,
"epoch": 1.0546875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.051196254789829254,
"kl": 0.01883502001874149,
"learning_rate": 4.12145522365689e-06,
"loss": 0.0001754522672854364,
"num_tokens": 650127.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 135,
"step_time": 5.736987513000258
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07493950799107552,
"epoch": 1.0625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007082380820065737,
"kl": 0.0019909495022147894,
"learning_rate": 4.104060653380403e-06,
"loss": 2.0618410417228006e-05,
"num_tokens": 654983.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 136,
"step_time": 6.561122189999878
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.125,
"completions/mean_terminated_length": 17.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07711902260780334,
"epoch": 1.0703125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.061376169323921204,
"kl": 0.021036528050899506,
"learning_rate": 4.086533074620919e-06,
"loss": 0.0002116656833095476,
"num_tokens": 659836.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 137,
"step_time": 6.81616649800003
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 27.0,
"completions/mean_terminated_length": 27.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.038892749696969986,
"epoch": 1.078125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005187615752220154,
"kl": 0.0026534507051110268,
"learning_rate": 4.068873940762796e-06,
"loss": 2.6154470106121153e-05,
"num_tokens": 664044.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 138,
"step_time": 5.881332467999982
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 15.625,
"completions/mean_terminated_length": 15.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1011301763355732,
"epoch": 1.0859375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.2945476174354553,
"kl": 0.08709336444735527,
"learning_rate": 4.051084716098921e-06,
"loss": 0.0008627058705314994,
"num_tokens": 668817.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 139,
"step_time": 6.639636123999935
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.625,
"completions/mean_terminated_length": 13.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12177715450525284,
"epoch": 1.09375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.033345598727464676,
"kl": 0.005186434369534254,
"learning_rate": 4.033166875709291e-06,
"loss": 5.185226473258808e-05,
"num_tokens": 674298.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 140,
"step_time": 5.517028535000009
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.5,
"completions/mean_terminated_length": 17.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07133128866553307,
"epoch": 1.1015625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.12929074466228485,
"kl": 0.09029890596866608,
"learning_rate": 4.015121905338704e-06,
"loss": 0.0009029890061356127,
"num_tokens": 678246.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 141,
"step_time": 5.957433755000011
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.057012878358364105,
"epoch": 1.109375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0789097398519516,
"kl": 0.03356556943617761,
"learning_rate": 3.996951301273556e-06,
"loss": 0.0003683689865283668,
"num_tokens": 682300.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 142,
"step_time": 5.954758885000047
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.125,
"completions/mean_terminated_length": 19.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07221270352602005,
"epoch": 1.1171875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.08531598001718521,
"kl": 0.04459764843340963,
"learning_rate": 3.9786565702177725e-06,
"loss": 0.00040393066592514515,
"num_tokens": 687145.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 143,
"step_time": 6.8996876880000855
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.25,
"completions/mean_terminated_length": 13.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.14312979578971863,
"epoch": 1.125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.020980026572942734,
"kl": 0.03795890283072367,
"learning_rate": 3.960239229167869e-06,
"loss": 0.0003864379250444472,
"num_tokens": 692651.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 144,
"step_time": 5.689572013000088
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.125,
"completions/mean_terminated_length": 13.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13573984801769257,
"epoch": 1.1328125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.014293034560978413,
"kl": 0.003559689619578421,
"learning_rate": 3.941700805287169e-06,
"loss": 3.578899850253947e-05,
"num_tokens": 698156.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 145,
"step_time": 5.635921549999921
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.75,
"completions/mean_terminated_length": 19.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07221874222159386,
"epoch": 1.140625,
"frac_reward_zero_std": 0.5,
"grad_norm": 2.1729769706726074,
"kl": 0.006373312848154455,
"learning_rate": 3.92304283577916e-06,
"loss": -0.15180744230747223,
"num_tokens": 703038.0,
"reward": 0.8812500238418579,
"reward_std": 0.3358757197856903,
"rewards/reward_fn/mean": 0.8812500238418579,
"rewards/reward_fn/std": 0.3358757197856903,
"step": 146,
"step_time": 6.94864533499981
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.625,
"completions/mean_terminated_length": 19.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06784151494503021,
"epoch": 1.1484375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008188405074179173,
"kl": 0.005511927942279726,
"learning_rate": 3.904266867760044e-06,
"loss": 5.116416286909953e-05,
"num_tokens": 707899.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 147,
"step_time": 6.646591747999992
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.5,
"completions/mean_terminated_length": 19.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07835190743207932,
"epoch": 1.15625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004332108423113823,
"kl": 0.0011774248559959233,
"learning_rate": 3.8853744581304376e-06,
"loss": 1.2425527529558167e-05,
"num_tokens": 712639.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 148,
"step_time": 6.73792784200009
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.375,
"completions/mean_terminated_length": 19.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07140998169779778,
"epoch": 1.1640625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006420095916837454,
"kl": 0.008202763739973307,
"learning_rate": 3.866367173446281e-06,
"loss": 7.896333409007639e-05,
"num_tokens": 717406.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 149,
"step_time": 6.581852653999931
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.125,
"completions/mean_terminated_length": 17.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07744136825203896,
"epoch": 1.171875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006486960221081972,
"kl": 0.0021612545242533088,
"learning_rate": 3.84724658978894e-06,
"loss": 2.1575528080575168e-05,
"num_tokens": 722111.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 150,
"step_time": 6.601302765000128
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.5,
"completions/mean_terminated_length": 13.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12781532481312752,
"epoch": 1.1796875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.011086604557931423,
"kl": 0.0017027303110808134,
"learning_rate": 3.828014292634508e-06,
"loss": 1.7027303329086863e-05,
"num_tokens": 727595.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 151,
"step_time": 5.656125526000324
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.75,
"completions/mean_terminated_length": 19.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07872535288333893,
"epoch": 1.1875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.030335335060954094,
"kl": 0.017175353597849607,
"learning_rate": 3.808671876722357e-06,
"loss": 0.00016482984938193113,
"num_tokens": 732313.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 152,
"step_time": 6.728792943000144
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 23.25,
"completions/mean_terminated_length": 23.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06778555735945702,
"epoch": 1.1953125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005629899445921183,
"kl": 0.001822992053348571,
"learning_rate": 3.7892209459228802e-06,
"loss": 1.8229919078294188e-05,
"num_tokens": 737211.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 153,
"step_time": 7.577010963999783
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 25.375,
"completions/mean_terminated_length": 25.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03097234107553959,
"epoch": 1.203125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00039684277726337314,
"kl": 0.0014657879946753383,
"learning_rate": 3.769663113104516e-06,
"loss": 1.4650184311904013e-05,
"num_tokens": 741306.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 154,
"step_time": 5.836096522999696
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.75,
"completions/mean_terminated_length": 13.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1351180225610733,
"epoch": 1.2109375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.04005980119109154,
"kl": 0.012403905857354403,
"learning_rate": 3.7500000000000005e-06,
"loss": 0.00012463353050407022,
"num_tokens": 746788.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 155,
"step_time": 5.403199547000213
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 25.25,
"completions/mean_terminated_length": 25.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05965032987296581,
"epoch": 1.21875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007664468139410019,
"kl": 0.0025948547408916056,
"learning_rate": 3.7302332370718988e-06,
"loss": 2.4826764274621382e-05,
"num_tokens": 751706.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 156,
"step_time": 7.5867298230000415
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 23.375,
"completions/mean_terminated_length": 23.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03286047466099262,
"epoch": 1.2265625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0014668001094833016,
"kl": 0.0018113566911779344,
"learning_rate": 3.7103644633774015e-06,
"loss": 1.7852235032478347e-05,
"num_tokens": 755809.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 157,
"step_time": 5.844333027999937
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.0,
"completions/mean_terminated_length": 17.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07944399863481522,
"epoch": 1.234375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00917236041277647,
"kl": 0.0026575025403872132,
"learning_rate": 3.690395326432421e-06,
"loss": 2.6575022275210358e-05,
"num_tokens": 760533.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 158,
"step_time": 6.835314456999868
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.375,
"completions/mean_terminated_length": 19.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07627736404538155,
"epoch": 1.2421875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.016258908435702324,
"kl": 0.003968668053857982,
"learning_rate": 3.6703274820749736e-06,
"loss": 3.887472485075705e-05,
"num_tokens": 765320.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 159,
"step_time": 6.874671181000167
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.375,
"completions/mean_terminated_length": 19.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06996462866663933,
"epoch": 1.25,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.011779413558542728,
"kl": 0.002461543306708336,
"learning_rate": 3.650162594327881e-06,
"loss": 2.4944250981207006e-05,
"num_tokens": 770199.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 160,
"step_time": 6.942916448000005
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 29.5,
"completions/mean_terminated_length": 29.5,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"entropy": 0.02681659162044525,
"epoch": 1.2578125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00040976243326440454,
"kl": 0.0014975924859754741,
"learning_rate": 3.6299023352607894e-06,
"loss": 1.4975924386817496e-05,
"num_tokens": 774395.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 161,
"step_time": 6.015989045999959
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 26.875,
"completions/mean_terminated_length": 26.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.061081623658537865,
"epoch": 1.265625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.014576679095625877,
"kl": 0.0038794887950643897,
"learning_rate": 3.6095483848515223e-06,
"loss": 3.555541479727253e-05,
"num_tokens": 779170.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 162,
"step_time": 7.245200251999904
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 22.125,
"completions/mean_terminated_length": 22.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06859908811748028,
"epoch": 1.2734375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.020452730357646942,
"kl": 0.005623190430924296,
"learning_rate": 3.589102430846773e-06,
"loss": 5.230373062659055e-05,
"num_tokens": 784051.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 163,
"step_time": 7.287538059000099
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 25.0,
"completions/mean_terminated_length": 25.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.028452389873564243,
"epoch": 1.28125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004462169948965311,
"kl": 0.001680202956777066,
"learning_rate": 3.5685661686221644e-06,
"loss": 1.6027215679059736e-05,
"num_tokens": 788131.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 164,
"step_time": 5.703693580999925
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.5,
"completions/mean_terminated_length": 19.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05600108206272125,
"epoch": 1.2890625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.021526971831917763,
"kl": 0.006003132904879749,
"learning_rate": 3.5479413010416606e-06,
"loss": 5.7515826483722776e-05,
"num_tokens": 792967.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 165,
"step_time": 6.575796513000114
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.5,
"completions/mean_terminated_length": 27.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.02504791971296072,
"epoch": 1.296875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0030265813693404198,
"kl": 0.0021526971249841154,
"learning_rate": 3.527229538316371e-06,
"loss": 2.115210190822836e-05,
"num_tokens": 797183.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 166,
"step_time": 5.972162173999777
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.375,
"completions/mean_terminated_length": 13.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13807643204927444,
"epoch": 1.3046875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03702490031719208,
"kl": 0.006319678155705333,
"learning_rate": 3.5064325978627365e-06,
"loss": 6.32827723165974e-05,
"num_tokens": 802710.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 167,
"step_time": 5.624996752000243
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.875,
"completions/mean_terminated_length": 27.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.02927943505346775,
"epoch": 1.3125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0027886745519936085,
"kl": 0.001402048219460994,
"learning_rate": 3.4855522041601265e-06,
"loss": 1.3811075405101292e-05,
"num_tokens": 806769.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 168,
"step_time": 5.729365316999974
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 22.875,
"completions/mean_terminated_length": 22.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09153914824128151,
"epoch": 1.3203125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.031587954610586166,
"kl": 0.006058453815057874,
"learning_rate": 3.4645900886078388e-06,
"loss": 6.320517422864214e-05,
"num_tokens": 812348.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 169,
"step_time": 7.63299943700008
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.5,
"completions/mean_terminated_length": 21.5,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"entropy": 0.05722755752503872,
"epoch": 1.328125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.014601275324821472,
"kl": 0.003756172431167215,
"learning_rate": 3.443547989381536e-06,
"loss": 3.436290717218071e-05,
"num_tokens": 817204.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 170,
"step_time": 6.65200465199996
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09964188188314438,
"epoch": 1.3359375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.06053476780653,
"kl": 0.006807157536968589,
"learning_rate": 3.422427651289118e-06,
"loss": 6.788011523894966e-05,
"num_tokens": 822734.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 171,
"step_time": 7.496788298999945
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 29.625,
"completions/mean_terminated_length": 29.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05451471544802189,
"epoch": 1.34375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008337209932506084,
"kl": 0.0015613330178894103,
"learning_rate": 3.4012308256260366e-06,
"loss": 1.5506457202718593e-05,
"num_tokens": 827671.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 172,
"step_time": 7.244489809999777
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 128.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 38.875,
"completions/mean_terminated_length": 26.142858505249023,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.6550341844558716,
"epoch": 1.3515625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008499568328261375,
"kl": 0.0018006233731284738,
"learning_rate": 3.3799592700300867e-06,
"loss": 1.8337512301513925e-05,
"num_tokens": 833354.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 173,
"step_time": 14.681682713999862
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 23.875,
"completions/mean_terminated_length": 23.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.060140494257211685,
"epoch": 1.359375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00928256381303072,
"kl": 0.0029123042477294803,
"learning_rate": 3.3586147483356534e-06,
"loss": 2.732695429585874e-05,
"num_tokens": 838109.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 174,
"step_time": 7.2412520599998516
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 40.0,
"completions/max_terminated_length": 40.0,
"completions/mean_length": 24.375,
"completions/mean_terminated_length": 24.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06309142522513866,
"epoch": 1.3671875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00517475139349699,
"kl": 0.0014793115551583469,
"learning_rate": 3.3371990304274654e-06,
"loss": 1.494552179792663e-05,
"num_tokens": 843028.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 175,
"step_time": 7.81271701799983
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.0,
"completions/mean_terminated_length": 21.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03291504085063934,
"epoch": 1.375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004775932524353266,
"kl": 0.002072959439828992,
"learning_rate": 3.315713892093829e-06,
"loss": 2.0729594325530343e-05,
"num_tokens": 847064.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 176,
"step_time": 5.776862065000159
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 35.0,
"completions/max_terminated_length": 35.0,
"completions/mean_length": 16.25,
"completions/mean_terminated_length": 16.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13138685747981071,
"epoch": 1.3828125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.023814553394913673,
"kl": 0.0031949164113029838,
"learning_rate": 3.294161114879382e-06,
"loss": 3.185410605510697e-05,
"num_tokens": 852594.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 177,
"step_time": 7.431254295000144
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 16.25,
"completions/mean_terminated_length": 16.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12037009745836258,
"epoch": 1.390625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.018406132236123085,
"kl": 0.003100107191130519,
"learning_rate": 3.272542485937369e-06,
"loss": 3.313878914923407e-05,
"num_tokens": 858124.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 178,
"step_time": 7.8139032770000085
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 39.0,
"completions/max_terminated_length": 39.0,
"completions/mean_length": 28.625,
"completions/mean_terminated_length": 28.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.16150045860558748,
"epoch": 1.3984375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.24875901639461517,
"kl": 0.045317163807339966,
"learning_rate": 3.2508597978814515e-06,
"loss": 0.00043976842425763607,
"num_tokens": 862249.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 179,
"step_time": 6.564737340999727
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.5,
"completions/mean_terminated_length": 27.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03088196087628603,
"epoch": 1.40625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0007541459635831416,
"kl": 0.0011516560916788876,
"learning_rate": 3.2291148486370626e-06,
"loss": 1.1446018106653355e-05,
"num_tokens": 866265.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 180,
"step_time": 5.712320224999758
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.75,
"completions/mean_terminated_length": 21.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06882932037115097,
"epoch": 1.4140625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00206294865347445,
"kl": 0.0013715573586523533,
"learning_rate": 3.207309441292325e-06,
"loss": 1.3715573004446924e-05,
"num_tokens": 871011.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 181,
"step_time": 6.726993576000041
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 90.0,
"completions/max_terminated_length": 90.0,
"completions/mean_length": 29.125,
"completions/mean_terminated_length": 29.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.5393692404031754,
"epoch": 1.421875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02723606675863266,
"kl": 0.0031313110375776887,
"learning_rate": 3.185445383948539e-06,
"loss": 3.407296753721312e-05,
"num_tokens": 876640.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 182,
"step_time": 11.819733140000153
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 20.75,
"completions/mean_terminated_length": 20.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12167311646044254,
"epoch": 1.4296875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02585308626294136,
"kl": 0.002457446011248976,
"learning_rate": 3.1635244895702527e-06,
"loss": 2.4284261598950252e-05,
"num_tokens": 881406.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 183,
"step_time": 6.903921601999855
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 23.0,
"completions/mean_terminated_length": 23.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06957883015275002,
"epoch": 1.4375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0076453592628240585,
"kl": 0.001370629877783358,
"learning_rate": 3.1415485758349344e-06,
"loss": 1.3765486073680222e-05,
"num_tokens": 886226.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 184,
"step_time": 7.334037624999837
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.5,
"completions/mean_terminated_length": 19.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0735643245279789,
"epoch": 1.4453125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005766255781054497,
"kl": 0.0016721467836759984,
"learning_rate": 3.11951946498225e-06,
"loss": 1.5945741324685514e-05,
"num_tokens": 890954.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 185,
"step_time": 6.7520039959997575
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 77.0,
"completions/max_terminated_length": 77.0,
"completions/mean_length": 33.375,
"completions/mean_terminated_length": 33.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.043096503242850304,
"epoch": 1.453125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004805476404726505,
"kl": 0.001356584718450904,
"learning_rate": 3.0974389836629628e-06,
"loss": 1.3770947589364368e-05,
"num_tokens": 895109.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 186,
"step_time": 9.572878103999756
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 16.375,
"completions/mean_terminated_length": 16.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1160760298371315,
"epoch": 1.4609375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.021316073834896088,
"kl": 0.0027994479751214385,
"learning_rate": 3.0753089627874668e-06,
"loss": 2.722722274484113e-05,
"num_tokens": 900640.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 187,
"step_time": 7.767691414999717
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 25.0,
"completions/mean_terminated_length": 25.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.034027645364403725,
"epoch": 1.46875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0012773608323186636,
"kl": 0.0015459859278053045,
"learning_rate": 3.0531312373739695e-06,
"loss": 1.545986015116796e-05,
"num_tokens": 904644.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 188,
"step_time": 5.866847418999896
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 91.0,
"completions/max_terminated_length": 91.0,
"completions/mean_length": 31.75,
"completions/mean_terminated_length": 31.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05926687642931938,
"epoch": 1.4765625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02531527541577816,
"kl": 0.0033996477723121643,
"learning_rate": 3.030907646396333e-06,
"loss": 3.2442520023323596e-05,
"num_tokens": 909574.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 189,
"step_time": 11.722558253999978
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 32.0,
"completions/max_terminated_length": 32.0,
"completions/mean_length": 21.875,
"completions/mean_terminated_length": 21.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06604529172182083,
"epoch": 1.484375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.014295347966253757,
"kl": 0.004741971264593303,
"learning_rate": 3.0086400326315853e-06,
"loss": 4.750135849462822e-05,
"num_tokens": 914337.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 190,
"step_time": 6.908794395000086
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.0,
"completions/mean_terminated_length": 21.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06935635954141617,
"epoch": 1.4921875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0014209807850420475,
"kl": 0.0011486579023767263,
"learning_rate": 2.9863302425071156e-06,
"loss": 1.2237365808687173e-05,
"num_tokens": 919185.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 191,
"step_time": 6.616372841999919
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 18.75,
"completions/mean_terminated_length": 18.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11195787042379379,
"epoch": 1.5,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.11325943470001221,
"kl": 0.004804897769645322,
"learning_rate": 2.963980125947573e-06,
"loss": 6.231457518879324e-05,
"num_tokens": 924735.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 192,
"step_time": 7.530646898999976
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 35.0,
"completions/max_terminated_length": 35.0,
"completions/mean_length": 15.875,
"completions/mean_terminated_length": 15.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1282949000597,
"epoch": 1.5078125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02611485868692398,
"kl": 0.001775389551767148,
"learning_rate": 2.941591536221469e-06,
"loss": 2.0562109057209454e-05,
"num_tokens": 930262.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 193,
"step_time": 7.432371278000119
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 27.0,
"completions/mean_terminated_length": 27.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.029940275475382805,
"epoch": 1.515625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0013159937225282192,
"kl": 0.001477666082791984,
"learning_rate": 2.9191663297875027e-06,
"loss": 1.4661342902400065e-05,
"num_tokens": 934362.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 194,
"step_time": 5.929965283999991
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.5,
"completions/mean_terminated_length": 13.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.142668217420578,
"epoch": 1.5234375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.05234305188059807,
"kl": 0.005082390503957868,
"learning_rate": 2.896706366140629e-06,
"loss": 5.082390271127224e-05,
"num_tokens": 939870.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 195,
"step_time": 5.839425092999818
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.125,
"completions/mean_terminated_length": 21.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06505489535629749,
"epoch": 1.53125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.002731472020968795,
"kl": 0.001581202377565205,
"learning_rate": 2.8742135076578608e-06,
"loss": 1.5823465219000354e-05,
"num_tokens": 944735.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 196,
"step_time": 6.905690054999923
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 19.375,
"completions/mean_terminated_length": 19.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11012165620923042,
"epoch": 1.5390625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0070010521449148655,
"kl": 0.0010562781826592982,
"learning_rate": 2.8516896194438515e-06,
"loss": 1.0670170013327152e-05,
"num_tokens": 950290.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 197,
"step_time": 7.652169488000027
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 35.0,
"completions/max_terminated_length": 35.0,
"completions/mean_length": 21.875,
"completions/mean_terminated_length": 21.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06521525606513023,
"epoch": 1.546875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.017679838463664055,
"kl": 0.0035706963972188532,
"learning_rate": 2.8291365691762313e-06,
"loss": 3.664140967885032e-05,
"num_tokens": 955021.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 198,
"step_time": 7.155417588999853
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 25.25,
"completions/mean_terminated_length": 25.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.049559252336621284,
"epoch": 1.5546875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.05125805735588074,
"kl": 0.01641816832125187,
"learning_rate": 2.8065562269507464e-06,
"loss": 0.00016517679614480585,
"num_tokens": 959027.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 199,
"step_time": 5.624229268999898
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 26.875,
"completions/mean_terminated_length": 26.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.04596647992730141,
"epoch": 1.5625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.009446537122130394,
"kl": 0.0029358353931456804,
"learning_rate": 2.7839504651261873e-06,
"loss": 3.024380566785112e-05,
"num_tokens": 963106.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 200,
"step_time": 5.7302313129998765
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 24.125,
"completions/mean_terminated_length": 24.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06624070927500725,
"epoch": 1.5703125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0009546372457407415,
"kl": 0.0009917069983202964,
"learning_rate": 2.761321158169134e-06,
"loss": 1.0749818102340214e-05,
"num_tokens": 968011.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 201,
"step_time": 7.5783325940001305
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 23.125,
"completions/mean_terminated_length": 23.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03553210385143757,
"epoch": 1.578125,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.7552725076675415,
"kl": 0.06742474652128294,
"learning_rate": 2.7386701824985257e-06,
"loss": -0.07762715220451355,
"num_tokens": 972088.0,
"reward": 0.8812500238418579,
"reward_std": 0.3358757197856903,
"rewards/reward_fn/mean": 0.8812500238418579,
"rewards/reward_fn/std": 0.3358757197856903,
"step": 202,
"step_time": 5.655839926999988
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.125,
"completions/mean_terminated_length": 21.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06611593812704086,
"epoch": 1.5859375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.002574133686721325,
"kl": 0.0014704846544191241,
"learning_rate": 2.715999416330068e-06,
"loss": 1.4717452359036542e-05,
"num_tokens": 976941.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 203,
"step_time": 6.626762887999803
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 16.375,
"completions/mean_terminated_length": 16.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12001239135861397,
"epoch": 1.59375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.022112533450126648,
"kl": 0.004337176156695932,
"learning_rate": 2.6933107395204926e-06,
"loss": 3.791246490436606e-05,
"num_tokens": 982472.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 204,
"step_time": 7.7095311099999435
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.125,
"completions/mean_terminated_length": 21.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.037428947165608406,
"epoch": 1.6015625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005030359141528606,
"kl": 0.002402382204309106,
"learning_rate": 2.670606033411678e-06,
"loss": 2.403911275905557e-05,
"num_tokens": 986525.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 205,
"step_time": 5.898442960000239
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 23.875,
"completions/mean_terminated_length": 23.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06124606728553772,
"epoch": 1.609375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.017411913722753525,
"kl": 0.004249897378031164,
"learning_rate": 2.6478871806746496e-06,
"loss": 3.866076440317556e-05,
"num_tokens": 991340.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 206,
"step_time": 7.1736011519999465
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 24.0,
"completions/mean_terminated_length": 24.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07086298987269402,
"epoch": 1.6171875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007288594264537096,
"kl": 0.0015791382757015526,
"learning_rate": 2.625156065153473e-06,
"loss": 1.550133674754761e-05,
"num_tokens": 996212.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 207,
"step_time": 7.194994807000057
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 18.875,
"completions/mean_terminated_length": 18.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.10177021846175194,
"epoch": 1.625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.012765838764607906,
"kl": 0.003196087433025241,
"learning_rate": 2.602414571709036e-06,
"loss": 3.193688462488353e-05,
"num_tokens": 1001759.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 208,
"step_time": 7.412187181000036
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.25,
"completions/mean_terminated_length": 13.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13615204393863678,
"epoch": 1.6328125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01755693182349205,
"kl": 0.0030227339593693614,
"learning_rate": 2.5796645860627665e-06,
"loss": 3.0503688321914524e-05,
"num_tokens": 1007265.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 209,
"step_time": 5.581415800000059
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 54.0,
"completions/max_terminated_length": 54.0,
"completions/mean_length": 28.75,
"completions/mean_terminated_length": 28.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.15063174068927765,
"epoch": 1.640625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007047915831208229,
"kl": 0.002616984653286636,
"learning_rate": 2.556907994640264e-06,
"loss": 2.5744397134985775e-05,
"num_tokens": 1011331.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 210,
"step_time": 7.517502090000107
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.375,
"completions/mean_terminated_length": 21.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05742892995476723,
"epoch": 1.6484375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006198030896484852,
"kl": 0.00281632284168154,
"learning_rate": 2.5341466844148775e-06,
"loss": 2.817007407429628e-05,
"num_tokens": 1016182.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 211,
"step_time": 6.626822569000069
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 67.0,
"completions/max_terminated_length": 67.0,
"completions/mean_length": 30.875,
"completions/mean_terminated_length": 30.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06470214761793613,
"epoch": 1.65625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007548889145255089,
"kl": 0.0024077071575447917,
"learning_rate": 2.511382542751239e-06,
"loss": 2.439593299641274e-05,
"num_tokens": 1021133.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 212,
"step_time": 9.792585082999722
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 128.0,
"completions/max_terminated_length": 107.0,
"completions/mean_length": 45.125,
"completions/mean_terminated_length": 33.28571701049805,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08313845098018646,
"epoch": 1.6640625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007846017368137836,
"kl": 0.0020210095099173486,
"learning_rate": 2.488617457248761e-06,
"loss": 1.9834918930428103e-05,
"num_tokens": 1026074.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 213,
"step_time": 14.740446427000279
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 25.0,
"completions/mean_terminated_length": 25.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03169822786003351,
"epoch": 1.671875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004874436184763908,
"kl": 0.0018856602837331593,
"learning_rate": 2.465853315585123e-06,
"loss": 1.8856600945582613e-05,
"num_tokens": 1030230.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 214,
"step_time": 5.843885092000164
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 41.0,
"completions/max_terminated_length": 41.0,
"completions/mean_length": 22.75,
"completions/mean_terminated_length": 22.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11479110643267632,
"epoch": 1.6796875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01489281002432108,
"kl": 0.003329504979774356,
"learning_rate": 2.443092005359736e-06,
"loss": 3.363801079103723e-05,
"num_tokens": 1034984.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 215,
"step_time": 7.630429413000456
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 23.25,
"completions/mean_terminated_length": 23.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06059539131820202,
"epoch": 1.6875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.011219343170523643,
"kl": 0.002763925353065133,
"learning_rate": 2.420335413937234e-06,
"loss": 2.7146501452079974e-05,
"num_tokens": 1039826.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 216,
"step_time": 7.484693694999805
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 24.125,
"completions/mean_terminated_length": 24.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06515759788453579,
"epoch": 1.6953125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006574144121259451,
"kl": 0.0019047525711357594,
"learning_rate": 2.3975854282909645e-06,
"loss": 1.847416569944471e-05,
"num_tokens": 1044679.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 217,
"step_time": 7.4298257559999
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 45.0,
"completions/max_terminated_length": 45.0,
"completions/mean_length": 27.0,
"completions/mean_terminated_length": 27.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05187015235424042,
"epoch": 1.703125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008155517280101776,
"kl": 0.002550492761656642,
"learning_rate": 2.374843934846528e-06,
"loss": 2.506848977645859e-05,
"num_tokens": 1048823.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 218,
"step_time": 7.006674997000118
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 15.875,
"completions/mean_terminated_length": 15.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12386251240968704,
"epoch": 1.7109375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.019120769575238228,
"kl": 0.003044891287572682,
"learning_rate": 2.35211281932535e-06,
"loss": 3.0685067031299695e-05,
"num_tokens": 1054326.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 219,
"step_time": 7.376969552999526
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.5,
"completions/mean_terminated_length": 13.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12342415750026703,
"epoch": 1.71875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.017717929556965828,
"kl": 0.004510085214860737,
"learning_rate": 2.3293939665883233e-06,
"loss": 4.5417702494887635e-05,
"num_tokens": 1059834.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 220,
"step_time": 5.555540719999954
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 40.0,
"completions/max_terminated_length": 40.0,
"completions/mean_length": 24.875,
"completions/mean_terminated_length": 24.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05634631961584091,
"epoch": 1.7265625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008950191549956799,
"kl": 0.0023297353181988,
"learning_rate": 2.306689260479508e-06,
"loss": 2.305710586369969e-05,
"num_tokens": 1064757.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 221,
"step_time": 7.691754480999407
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.125,
"completions/mean_terminated_length": 21.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06376663595438004,
"epoch": 1.734375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007804343942552805,
"kl": 0.0021846327581442893,
"learning_rate": 2.284000583669933e-06,
"loss": 2.1814143110532314e-05,
"num_tokens": 1069550.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 222,
"step_time": 6.525545030000103
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 90.0,
"completions/max_terminated_length": 90.0,
"completions/mean_length": 29.0,
"completions/mean_terminated_length": 29.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08469109237194061,
"epoch": 1.7421875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008673850446939468,
"kl": 0.0024135791463777423,
"learning_rate": 2.261329817501475e-06,
"loss": 2.4780489184195176e-05,
"num_tokens": 1074362.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 223,
"step_time": 11.37344323599973
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07181186601519585,
"epoch": 1.75,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00881747156381607,
"kl": 0.0025651361793279648,
"learning_rate": 2.238678841830867e-06,
"loss": 2.4443201255053282e-05,
"num_tokens": 1079154.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 224,
"step_time": 6.7924440969995885
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.0,
"completions/mean_terminated_length": 21.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07576226070523262,
"epoch": 1.7578125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.009993416257202625,
"kl": 0.0031048988457769156,
"learning_rate": 2.2160495348738127e-06,
"loss": 3.1577099434798583e-05,
"num_tokens": 1084034.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 225,
"step_time": 6.582048697000118
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06813675723969936,
"epoch": 1.765625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.013243633322417736,
"kl": 0.0029321471229195595,
"learning_rate": 2.1934437730492544e-06,
"loss": 2.889215829782188e-05,
"num_tokens": 1088838.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 226,
"step_time": 6.774607225000182
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 96.0,
"completions/max_terminated_length": 96.0,
"completions/mean_length": 27.625,
"completions/mean_terminated_length": 27.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0714375339448452,
"epoch": 1.7734375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006967134308069944,
"kl": 0.0022760473075322807,
"learning_rate": 2.1708634308237687e-06,
"loss": 2.1496147383004427e-05,
"num_tokens": 1093735.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 227,
"step_time": 11.807745952999994
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 31.0,
"completions/max_terminated_length": 31.0,
"completions/mean_length": 19.375,
"completions/mean_terminated_length": 19.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11793160066008568,
"epoch": 1.78125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.05189337208867073,
"kl": 0.014766670181415975,
"learning_rate": 2.1483103805561493e-06,
"loss": 0.00013642504927702248,
"num_tokens": 1098522.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 228,
"step_time": 6.672160937999706
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06688312441110611,
"epoch": 1.7890625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.011055883020162582,
"kl": 0.0036998813739046454,
"learning_rate": 2.1257864923421405e-06,
"loss": 3.6064928281120956e-05,
"num_tokens": 1103364.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 229,
"step_time": 6.56204488100002
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 114.0,
"completions/max_terminated_length": 114.0,
"completions/mean_length": 33.875,
"completions/mean_terminated_length": 33.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.051064563915133476,
"epoch": 1.796875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008055821992456913,
"kl": 0.0023585216840729117,
"learning_rate": 2.1032936338593716e-06,
"loss": 2.3143395083025098e-05,
"num_tokens": 1107471.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 230,
"step_time": 12.61657659399998
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 83.0,
"completions/max_terminated_length": 83.0,
"completions/mean_length": 25.875,
"completions/mean_terminated_length": 25.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0740746557712555,
"epoch": 1.8046875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007505510468035936,
"kl": 0.0018645531381480396,
"learning_rate": 2.080833670212498e-06,
"loss": 1.8865794118028134e-05,
"num_tokens": 1112306.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 231,
"step_time": 11.113060962999953
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.125,
"completions/mean_terminated_length": 21.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0632933359593153,
"epoch": 1.8125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005921315401792526,
"kl": 0.0018682752852328122,
"learning_rate": 2.0584084637785316e-06,
"loss": 1.870766755018849e-05,
"num_tokens": 1117095.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 232,
"step_time": 6.704811484000402
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 40.0,
"completions/max_terminated_length": 40.0,
"completions/mean_length": 23.875,
"completions/mean_terminated_length": 23.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08165674656629562,
"epoch": 1.8203125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.012375043705105782,
"kl": 0.0033231035922653973,
"learning_rate": 2.036019874052428e-06,
"loss": 3.117723827017471e-05,
"num_tokens": 1121946.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 233,
"step_time": 7.801074556999993
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 31.0,
"completions/max_terminated_length": 31.0,
"completions/mean_length": 19.5,
"completions/mean_terminated_length": 19.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08600634709000587,
"epoch": 1.828125,
"frac_reward_zero_std": 0.5,
"grad_norm": 1.3644095659255981,
"kl": 0.06359560182318091,
"learning_rate": 2.0136697574928853e-06,
"loss": 0.06466268002986908,
"num_tokens": 1126706.0,
"reward": 0.875,
"reward_std": 0.3535533845424652,
"rewards/reward_fn/mean": 0.875,
"rewards/reward_fn/std": 0.3535533845424652,
"step": 234,
"step_time": 6.896724009000081
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 43.0,
"completions/max_terminated_length": 43.0,
"completions/mean_length": 25.375,
"completions/mean_terminated_length": 25.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08186899498105049,
"epoch": 1.8359375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.011324622668325901,
"kl": 0.003795566619373858,
"learning_rate": 1.991359967368416e-06,
"loss": 3.484051558189094e-05,
"num_tokens": 1131529.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 235,
"step_time": 7.659685747999902
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.125,
"completions/mean_terminated_length": 17.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.072533143684268,
"epoch": 1.84375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.015638258308172226,
"kl": 0.003511433256790042,
"learning_rate": 1.9690923536036673e-06,
"loss": 3.5228476917836815e-05,
"num_tokens": 1136314.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 236,
"step_time": 6.76509731599981
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.0,
"completions/mean_terminated_length": 21.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0642678551375866,
"epoch": 1.8515625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006433664355427027,
"kl": 0.0019541659858077765,
"learning_rate": 1.9468687626260314e-06,
"loss": 1.9541659639799036e-05,
"num_tokens": 1141102.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 237,
"step_time": 6.493211311999858
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 23.0,
"completions/mean_terminated_length": 23.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03841168060898781,
"epoch": 1.859375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.014958206564188004,
"kl": 0.004870379460044205,
"learning_rate": 1.9246910372125345e-06,
"loss": 4.770414670929313e-05,
"num_tokens": 1145026.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 238,
"step_time": 5.479256311999961
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 40.0,
"completions/max_terminated_length": 40.0,
"completions/mean_length": 16.75,
"completions/mean_terminated_length": 16.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11552144214510918,
"epoch": 1.8671875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.06090380623936653,
"kl": 0.012604215648025274,
"learning_rate": 1.9025610163370385e-06,
"loss": 0.0001225811429321766,
"num_tokens": 1150584.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 239,
"step_time": 7.893678880999687
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.0,
"completions/mean_terminated_length": 21.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06876265630126,
"epoch": 1.875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.009387739934027195,
"kl": 0.003080976544879377,
"learning_rate": 1.8804805350177507e-06,
"loss": 2.9121343686711043e-05,
"num_tokens": 1155364.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 240,
"step_time": 6.4674589349997404
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.375,
"completions/mean_terminated_length": 13.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.14056620001792908,
"epoch": 1.8828125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.051601000130176544,
"kl": 0.009546731133013964,
"learning_rate": 1.8584514241650667e-06,
"loss": 9.567832603352144e-05,
"num_tokens": 1160847.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 241,
"step_time": 5.5828178320002735
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.875,
"completions/mean_terminated_length": 27.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.036137545481324196,
"epoch": 1.890625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.009184165857732296,
"kl": 0.0031361767323687673,
"learning_rate": 1.8364755104297477e-06,
"loss": 3.063581243623048e-05,
"num_tokens": 1165038.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 242,
"step_time": 5.927890447000209
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 40.0,
"completions/max_terminated_length": 40.0,
"completions/mean_length": 22.75,
"completions/mean_terminated_length": 22.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07793265208601952,
"epoch": 1.8984375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02107059955596924,
"kl": 0.006186584243550897,
"learning_rate": 1.8145546160514622e-06,
"loss": 6.0147060139570385e-05,
"num_tokens": 1169948.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 243,
"step_time": 7.625320269999975
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 13.0,
"completions/max_terminated_length": 13.0,
"completions/mean_length": 13.0,
"completions/mean_terminated_length": 13.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12505637109279633,
"epoch": 1.90625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.020931562408804893,
"kl": 0.004113408271223307,
"learning_rate": 1.792690558707675e-06,
"loss": 4.1134080674964935e-05,
"num_tokens": 1175424.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 244,
"step_time": 5.521570587000042
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.25,
"completions/mean_terminated_length": 21.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.043781496584415436,
"epoch": 1.9140625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.025867925956845284,
"kl": 0.010342990979552269,
"learning_rate": 1.7708851513629376e-06,
"loss": 9.541432518744841e-05,
"num_tokens": 1179346.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 245,
"step_time": 5.745181904000219
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 20.875,
"completions/mean_terminated_length": 20.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07973140105605125,
"epoch": 1.921875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01369334477931261,
"kl": 0.004745342535898089,
"learning_rate": 1.7491402021185489e-06,
"loss": 4.806690776604228e-05,
"num_tokens": 1184229.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 246,
"step_time": 6.7438787659998525
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 128.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 30.625,
"completions/mean_terminated_length": 16.71428680419922,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09480598196387291,
"epoch": 1.9296875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.04195118695497513,
"kl": 0.008697986835613847,
"learning_rate": 1.7274575140626318e-06,
"loss": 7.138060755096376e-05,
"num_tokens": 1189874.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 247,
"step_time": 15.305213977000221
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.375,
"completions/mean_terminated_length": 13.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12500061094760895,
"epoch": 1.9375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0684664249420166,
"kl": 0.013888376764953136,
"learning_rate": 1.7058388851206187e-06,
"loss": 0.00013920964556746185,
"num_tokens": 1195381.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 248,
"step_time": 5.70922280700006
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 25.375,
"completions/mean_terminated_length": 25.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03869021311402321,
"epoch": 1.9453125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.015580818057060242,
"kl": 0.004994981922209263,
"learning_rate": 1.6842861079061717e-06,
"loss": 4.9939146265387535e-05,
"num_tokens": 1199568.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 249,
"step_time": 6.0302398560002075
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.875,
"completions/mean_terminated_length": 27.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03580416925251484,
"epoch": 1.953125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.012648792937397957,
"kl": 0.004584258887916803,
"learning_rate": 1.6628009695725348e-06,
"loss": 4.5331114961300045e-05,
"num_tokens": 1203687.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 250,
"step_time": 6.181951604999995
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.25,
"completions/mean_terminated_length": 13.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11927280202507973,
"epoch": 1.9609375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.06169552356004715,
"kl": 0.016291253734380007,
"learning_rate": 1.6413852516643468e-06,
"loss": 0.0001629125326871872,
"num_tokens": 1209189.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 251,
"step_time": 5.695720361999975
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 37.0,
"completions/max_terminated_length": 37.0,
"completions/mean_length": 16.125,
"completions/mean_terminated_length": 16.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.10105976462364197,
"epoch": 1.96875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.04228847101330757,
"kl": 0.010480482131242752,
"learning_rate": 1.6200407299699141e-06,
"loss": 9.954295819625258e-05,
"num_tokens": 1214742.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 252,
"step_time": 7.750054464999721
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.04526445455849171,
"epoch": 1.9765625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03174661472439766,
"kl": 0.011470308061689138,
"learning_rate": 1.5987691743739636e-06,
"loss": 0.00011257198639214039,
"num_tokens": 1218788.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 253,
"step_time": 6.203933252000752
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.125,
"completions/mean_terminated_length": 13.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12663870304822922,
"epoch": 1.984375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02504121884703636,
"kl": 0.005651417304761708,
"learning_rate": 1.5775723487108821e-06,
"loss": 5.668877565767616e-05,
"num_tokens": 1224269.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 254,
"step_time": 5.6966200050001135
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.375,
"completions/mean_terminated_length": 19.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06413119472563267,
"epoch": 1.9921875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.036313824355602264,
"kl": 0.00865871855057776,
"learning_rate": 1.5564520106184643e-06,
"loss": 8.569318742956966e-05,
"num_tokens": 1229052.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 255,
"step_time": 6.825095515999692
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 15.0,
"completions/mean_terminated_length": 15.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09427187219262123,
"epoch": 2.0,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.028769060969352722,
"kl": 0.010795064270496368,
"learning_rate": 1.5354099113921614e-06,
"loss": 0.0001041956347762607,
"num_tokens": 1233740.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 256,
"step_time": 6.602496646999953
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 22.25,
"completions/mean_terminated_length": 22.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0705322865396738,
"epoch": 2.0078125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01884527876973152,
"kl": 0.004939868813380599,
"learning_rate": 1.514447795839874e-06,
"loss": 4.6844143071211874e-05,
"num_tokens": 1238474.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 257,
"step_time": 7.474117505999857
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.125,
"completions/mean_terminated_length": 17.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07266378402709961,
"epoch": 2.015625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.022645561024546623,
"kl": 0.0061811115592718124,
"learning_rate": 1.493567402137263e-06,
"loss": 6.193597073433921e-05,
"num_tokens": 1243295.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 258,
"step_time": 6.768364107999787
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 16.0,
"completions/mean_terminated_length": 16.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12317134439945221,
"epoch": 2.0234375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.027730176225304604,
"kl": 0.00469887419603765,
"learning_rate": 1.4727704616836297e-06,
"loss": 4.60884184576571e-05,
"num_tokens": 1248799.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 259,
"step_time": 7.577763272000084
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.25,
"completions/mean_terminated_length": 17.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07146281003952026,
"epoch": 2.03125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.032720040529966354,
"kl": 0.007956057321280241,
"learning_rate": 1.4520586989583406e-06,
"loss": 7.956057379487902e-05,
"num_tokens": 1253513.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 260,
"step_time": 6.917914829999518
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.75,
"completions/mean_terminated_length": 21.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06579671613872051,
"epoch": 2.0390625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02816769853234291,
"kl": 0.008500087773427367,
"learning_rate": 1.431433831377836e-06,
"loss": 7.657032983843237e-05,
"num_tokens": 1258259.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 261,
"step_time": 6.894851472000482
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.25,
"completions/mean_terminated_length": 13.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13017303496599197,
"epoch": 2.046875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.027672773227095604,
"kl": 0.005289856111630797,
"learning_rate": 1.4108975691532273e-06,
"loss": 5.289855835144408e-05,
"num_tokens": 1263765.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 262,
"step_time": 5.753936429000078
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.375,
"completions/mean_terminated_length": 19.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07268621399998665,
"epoch": 2.0546875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007546247914433479,
"kl": 0.002283608540892601,
"learning_rate": 1.3904516151484794e-06,
"loss": 2.186072015319951e-05,
"num_tokens": 1268632.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 263,
"step_time": 6.862648165999872
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 37.0,
"completions/max_terminated_length": 37.0,
"completions/mean_length": 16.0,
"completions/mean_terminated_length": 16.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11497627571225166,
"epoch": 2.0625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.013943173922598362,
"kl": 0.002583169029094279,
"learning_rate": 1.370097664739212e-06,
"loss": 2.6390163839096203e-05,
"num_tokens": 1274160.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 264,
"step_time": 7.761295357000108
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 16.375,
"completions/mean_terminated_length": 16.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11228630319237709,
"epoch": 2.0703125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02251162752509117,
"kl": 0.003555023460648954,
"learning_rate": 1.3498374056721198e-06,
"loss": 3.469496368779801e-05,
"num_tokens": 1279691.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 265,
"step_time": 7.840798843000357
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.0,
"completions/mean_terminated_length": 21.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06668232567608356,
"epoch": 2.078125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005440605338662863,
"kl": 0.0017036688514053822,
"learning_rate": 1.3296725179250274e-06,
"loss": 1.7276026483159512e-05,
"num_tokens": 1284475.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 266,
"step_time": 6.7484774439999455
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 18.125,
"completions/mean_terminated_length": 18.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08780457451939583,
"epoch": 2.0859375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02105136401951313,
"kl": 0.006752714281901717,
"learning_rate": 1.3096046735675795e-06,
"loss": 6.217646296136081e-05,
"num_tokens": 1289212.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 267,
"step_time": 7.876546445999793
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 23.0,
"completions/mean_terminated_length": 23.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03588521108031273,
"epoch": 2.09375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0117345554754138,
"kl": 0.003772490192204714,
"learning_rate": 1.2896355366226e-06,
"loss": 3.7270961911417544e-05,
"num_tokens": 1293364.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 268,
"step_time": 5.872368308999739
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.5,
"completions/mean_terminated_length": 19.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06518647447228432,
"epoch": 2.1015625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.021700672805309296,
"kl": 0.004600343410857022,
"learning_rate": 1.2697667629281025e-06,
"loss": 4.466335667530075e-05,
"num_tokens": 1298152.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 269,
"step_time": 6.83284532000016
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.5,
"completions/mean_terminated_length": 27.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03484191186726093,
"epoch": 2.109375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0065823025070130825,
"kl": 0.002110914676450193,
"learning_rate": 1.2500000000000007e-06,
"loss": 2.08322726393817e-05,
"num_tokens": 1302192.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 270,
"step_time": 5.96173259499983
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 16.25,
"completions/mean_terminated_length": 16.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12080564349889755,
"epoch": 2.1171875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01397998258471489,
"kl": 0.0026146003510802984,
"learning_rate": 1.2303368868954848e-06,
"loss": 2.3293352569453418e-05,
"num_tokens": 1307722.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 271,
"step_time": 7.747071295000296
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 15.125,
"completions/mean_terminated_length": 15.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09482500702142715,
"epoch": 2.125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.014719804748892784,
"kl": 0.0029105256544426084,
"learning_rate": 1.2107790540771208e-06,
"loss": 3.1068855605553836e-05,
"num_tokens": 1312491.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 272,
"step_time": 6.803867777999585
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.0,
"completions/mean_terminated_length": 17.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08690010197460651,
"epoch": 2.1328125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.012722927145659924,
"kl": 0.0026088722806889564,
"learning_rate": 1.1913281232776445e-06,
"loss": 3.0089617212070152e-05,
"num_tokens": 1317183.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 273,
"step_time": 6.73350407099997
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 18.875,
"completions/mean_terminated_length": 18.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.10723384842276573,
"epoch": 2.140625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.012960345484316349,
"kl": 0.001953385421074927,
"learning_rate": 1.1719857073654923e-06,
"loss": 1.958140819624532e-05,
"num_tokens": 1322710.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 274,
"step_time": 7.730757156000436
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 29.0,
"completions/mean_terminated_length": 29.0,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"entropy": 0.03018476441502571,
"epoch": 2.1484375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0026580565609037876,
"kl": 0.0016376653802581131,
"learning_rate": 1.1527534102110613e-06,
"loss": 1.6376652638427913e-05,
"num_tokens": 1326802.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 275,
"step_time": 5.899295565000557
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.375,
"completions/mean_terminated_length": 21.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.059426622465252876,
"epoch": 2.15625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.025692613795399666,
"kl": 0.007004954153671861,
"learning_rate": 1.1336328265537195e-06,
"loss": 7.027122774161398e-05,
"num_tokens": 1331677.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 276,
"step_time": 6.831976745000247
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 80.0,
"completions/max_terminated_length": 80.0,
"completions/mean_length": 35.375,
"completions/mean_terminated_length": 35.375,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"entropy": 0.03917406965047121,
"epoch": 2.1640625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0032123220153152943,
"kl": 0.001322296040598303,
"learning_rate": 1.1146255418695635e-06,
"loss": 1.3043942090007477e-05,
"num_tokens": 1335820.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 277,
"step_time": 10.114057109999976
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 24.625,
"completions/mean_terminated_length": 24.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.04752049781382084,
"epoch": 2.171875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.010773031041026115,
"kl": 0.004988847824279219,
"learning_rate": 1.0957331322399575e-06,
"loss": 4.370003443909809e-05,
"num_tokens": 1339885.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 278,
"step_time": 5.81339948699997
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 92.0,
"completions/max_terminated_length": 92.0,
"completions/mean_length": 28.875,
"completions/mean_terminated_length": 28.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07032647728919983,
"epoch": 2.1796875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005957606248557568,
"kl": 0.0015352739137597382,
"learning_rate": 1.0769571642208404e-06,
"loss": 1.4340959751280025e-05,
"num_tokens": 1344764.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 279,
"step_time": 11.989179764000255
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 37.0,
"completions/max_terminated_length": 37.0,
"completions/mean_length": 20.0,
"completions/mean_terminated_length": 20.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.15787944197654724,
"epoch": 2.1875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.022516168653964996,
"kl": 0.006525776581838727,
"learning_rate": 1.0582991947128324e-06,
"loss": 6.593053694814444e-05,
"num_tokens": 1348672.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 280,
"step_time": 6.449207413999829
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 17.25,
"completions/mean_terminated_length": 17.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0816731434315443,
"epoch": 2.1953125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.010547587648034096,
"kl": 0.002592157106846571,
"learning_rate": 1.0397607708321302e-06,
"loss": 2.5727105821715668e-05,
"num_tokens": 1353414.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 281,
"step_time": 7.077126720000251
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 25.375,
"completions/mean_terminated_length": 25.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.033613706938922405,
"epoch": 2.203125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0056767649948596954,
"kl": 0.0027879257686436176,
"learning_rate": 1.0213434297822275e-06,
"loss": 2.6398176487418823e-05,
"num_tokens": 1357505.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 282,
"step_time": 5.923250760999508
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 26.875,
"completions/mean_terminated_length": 26.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03769804444164038,
"epoch": 2.2109375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005933691281825304,
"kl": 0.00262093311175704,
"learning_rate": 1.0030486987264436e-06,
"loss": 2.5216926587745547e-05,
"num_tokens": 1361644.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 283,
"step_time": 5.9830027580001115
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 17.875,
"completions/mean_terminated_length": 17.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0936516635119915,
"epoch": 2.21875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.015039868652820587,
"kl": 0.003393694059923291,
"learning_rate": 9.848780946612962e-07,
"loss": 3.5618319088825956e-05,
"num_tokens": 1366347.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 284,
"step_time": 7.200875815000018
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06426311284303665,
"epoch": 2.2265625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007608731277287006,
"kl": 0.001917863031849265,
"learning_rate": 9.66833124290709e-07,
"loss": 1.9463377611828037e-05,
"num_tokens": 1371241.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 285,
"step_time": 6.8967751259997385
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 94.0,
"completions/max_terminated_length": 94.0,
"completions/mean_length": 27.25,
"completions/mean_terminated_length": 27.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06446886621415615,
"epoch": 2.234375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.010835153982043266,
"kl": 0.00325047189835459,
"learning_rate": 9.489152839010799e-07,
"loss": 2.8311722417129204e-05,
"num_tokens": 1376131.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 286,
"step_time": 11.860804265999832
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 17.5,
"completions/mean_terminated_length": 17.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08248795196413994,
"epoch": 2.2421875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.010750790126621723,
"kl": 0.0020258878357708454,
"learning_rate": 9.311260592372045e-07,
"loss": 2.2169147996464744e-05,
"num_tokens": 1380907.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 287,
"step_time": 6.986179221999919
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.5,
"completions/mean_terminated_length": 19.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11641774699091911,
"epoch": 2.25,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.011737891472876072,
"kl": 0.003091278951615095,
"learning_rate": 9.134669253790814e-07,
"loss": 2.849579141184222e-05,
"num_tokens": 1385623.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 288,
"step_time": 6.810695373999806
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.125,
"completions/mean_terminated_length": 21.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06711885333061218,
"epoch": 2.2578125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007158924825489521,
"kl": 0.0025333911762572825,
"learning_rate": 8.959393466195973e-07,
"loss": 2.5275945517932996e-05,
"num_tokens": 1390344.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 289,
"step_time": 6.85957102600014
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 17.25,
"completions/mean_terminated_length": 17.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07853703573346138,
"epoch": 2.265625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01479937881231308,
"kl": 0.004124688799493015,
"learning_rate": 8.785447763431101e-07,
"loss": 4.1246887121815234e-05,
"num_tokens": 1395054.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 290,
"step_time": 6.63891823799986
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.125,
"completions/max_length": 128.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 40.75,
"completions/mean_terminated_length": 28.285715103149414,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0656740702688694,
"epoch": 2.2734375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0060638003051280975,
"kl": 0.0017024054541252553,
"learning_rate": 8.612846569049324e-07,
"loss": 1.726844857330434e-05,
"num_tokens": 1400052.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 291,
"step_time": 14.556338565999795
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 23.375,
"completions/mean_terminated_length": 23.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03545568138360977,
"epoch": 2.28125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007708441931754351,
"kl": 0.0022156074410304427,
"learning_rate": 8.441604195117315e-07,
"loss": 2.2025331418262795e-05,
"num_tokens": 1404051.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 292,
"step_time": 5.789175566999802
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.125,
"completions/mean_terminated_length": 17.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.09139403142035007,
"epoch": 2.2890625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007536021526902914,
"kl": 0.0019620120874606073,
"learning_rate": 8.271734841028553e-07,
"loss": 1.879773844848387e-05,
"num_tokens": 1408800.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 293,
"step_time": 6.592784782999843
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.5,
"completions/mean_terminated_length": 21.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03953526355326176,
"epoch": 2.296875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01195154432207346,
"kl": 0.003599834628403187,
"learning_rate": 8.103252592325897e-07,
"loss": 3.324213685118593e-05,
"num_tokens": 1412752.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 294,
"step_time": 5.7077796439998565
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.125,
"completions/mean_terminated_length": 17.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08125988394021988,
"epoch": 2.3046875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007471000775694847,
"kl": 0.00221208727452904,
"learning_rate": 7.936171419533653e-07,
"loss": 2.2959266061661765e-05,
"num_tokens": 1417441.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 295,
"step_time": 6.576927980000164
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1136050745844841,
"epoch": 2.3125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.011644271202385426,
"kl": 0.0031743868021294475,
"learning_rate": 7.770505176999066e-07,
"loss": 2.477930684108287e-05,
"num_tokens": 1422995.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 296,
"step_time": 7.677933532000225
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.625,
"completions/mean_terminated_length": 19.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07930410280823708,
"epoch": 2.3203125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006496694404631853,
"kl": 0.0013743448653258383,
"learning_rate": 7.606267601743614e-07,
"loss": 1.3506290997611359e-05,
"num_tokens": 1427804.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 297,
"step_time": 6.850105020000228
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1292087510228157,
"epoch": 2.328125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02982700616121292,
"kl": 0.002042026782874018,
"learning_rate": 7.443472312323824e-07,
"loss": 1.8672115402296185e-05,
"num_tokens": 1433332.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 298,
"step_time": 7.489191680000204
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.375,
"completions/mean_terminated_length": 21.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05677143856883049,
"epoch": 2.3359375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007880290038883686,
"kl": 0.0026660520816221833,
"learning_rate": 7.282132807702144e-07,
"loss": 2.6689433070714585e-05,
"num_tokens": 1438211.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 299,
"step_time": 6.8618144679999205
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.5,
"completions/mean_terminated_length": 27.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0277756555005908,
"epoch": 2.34375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0013915153685957193,
"kl": 0.001870743464678526,
"learning_rate": 7.122262466127513e-07,
"loss": 1.8572343833511695e-05,
"num_tokens": 1442435.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 300,
"step_time": 6.021347086999867
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 107.0,
"completions/max_terminated_length": 107.0,
"completions/mean_length": 37.0,
"completions/mean_terminated_length": 37.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05828935466706753,
"epoch": 2.3515625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004984916187822819,
"kl": 0.0012245121179148555,
"learning_rate": 6.963874544026109e-07,
"loss": 1.2662912922678515e-05,
"num_tokens": 1447379.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 301,
"step_time": 13.41408500999978
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 16.25,
"completions/mean_terminated_length": 16.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12101850658655167,
"epoch": 2.359375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.06349971890449524,
"kl": 0.013034343253821135,
"learning_rate": 6.806982174902065e-07,
"loss": 0.00013196206418797374,
"num_tokens": 1452909.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 302,
"step_time": 7.468966810999973
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 29.5,
"completions/mean_terminated_length": 29.5,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"entropy": 0.02912551909685135,
"epoch": 2.3671875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0009350188192911446,
"kl": 0.0012789619504474103,
"learning_rate": 6.651598368248494e-07,
"loss": 1.2786756997229531e-05,
"num_tokens": 1456989.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 303,
"step_time": 5.737073586000406
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 108.0,
"completions/max_terminated_length": 108.0,
"completions/mean_length": 28.875,
"completions/mean_terminated_length": 28.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08968518674373627,
"epoch": 2.375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.010912664234638214,
"kl": 0.002259014407172799,
"learning_rate": 6.497736008468703e-07,
"loss": 2.482989293639548e-05,
"num_tokens": 1461908.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 304,
"step_time": 13.122933298000135
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.625,
"completions/mean_terminated_length": 19.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06897314265370369,
"epoch": 2.3828125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0196547731757164,
"kl": 0.006106121756602079,
"learning_rate": 6.345407853807864e-07,
"loss": 5.6373894040007144e-05,
"num_tokens": 1466729.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 305,
"step_time": 7.000840238999899
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.625,
"completions/mean_terminated_length": 13.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13361556828022003,
"epoch": 2.390625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02360568754374981,
"kl": 0.0038382920320145786,
"learning_rate": 6.194626535295059e-07,
"loss": 3.8651305658277124e-05,
"num_tokens": 1472214.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 306,
"step_time": 5.796944548999818
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07942482270300388,
"epoch": 2.3984375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004173581022769213,
"kl": 0.0015490494552068412,
"learning_rate": 6.045404555695935e-07,
"loss": 1.529452310933266e-05,
"num_tokens": 1476946.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 307,
"step_time": 6.881738667000263
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 40.0,
"completions/max_terminated_length": 40.0,
"completions/mean_length": 16.625,
"completions/mean_terminated_length": 16.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11725720018148422,
"epoch": 2.40625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.027200696989893913,
"kl": 0.006251339975278825,
"learning_rate": 5.897754288475979e-07,
"loss": 5.222540858085267e-05,
"num_tokens": 1482503.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 308,
"step_time": 7.865210730999479
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.375,
"completions/mean_terminated_length": 27.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03089694306254387,
"epoch": 2.4140625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.001481092651374638,
"kl": 0.0015604888903908432,
"learning_rate": 5.751687976774523e-07,
"loss": 1.5621018974343315e-05,
"num_tokens": 1486670.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 309,
"step_time": 5.85635403499964
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.25,
"completions/mean_terminated_length": 21.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07554454170167446,
"epoch": 2.421875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005015636794269085,
"kl": 0.001698852691333741,
"learning_rate": 5.607217732389503e-07,
"loss": 1.6451504052383825e-05,
"num_tokens": 1491408.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 310,
"step_time": 6.528806915999667
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 21.75,
"completions/mean_terminated_length": 21.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12106388807296753,
"epoch": 2.4296875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02396441251039505,
"kl": 0.0038617003010585904,
"learning_rate": 5.464355534773217e-07,
"loss": 2.8673672204604372e-05,
"num_tokens": 1496958.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 311,
"step_time": 7.533618949999891
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 18.875,
"completions/mean_terminated_length": 18.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11433938518166542,
"epoch": 2.4375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.003641946241259575,
"kl": 0.001054832770023495,
"learning_rate": 5.323113230038899e-07,
"loss": 9.623323421692476e-06,
"num_tokens": 1502481.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 312,
"step_time": 7.2610713440003565
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 25.0,
"completions/mean_terminated_length": 25.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0287599079310894,
"epoch": 2.4453125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.002698131836950779,
"kl": 0.0020212933886796236,
"learning_rate": 5.183502529978548e-07,
"loss": 2.0212932213325985e-05,
"num_tokens": 1506701.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 313,
"step_time": 5.806462265000391
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 25.375,
"completions/mean_terminated_length": 25.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.032324470579624176,
"epoch": 2.453125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0024203648790717125,
"kl": 0.0017645764746703207,
"learning_rate": 5.045535011091693e-07,
"loss": 1.707239425741136e-05,
"num_tokens": 1510856.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 314,
"step_time": 5.994479979000062
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 17.25,
"completions/mean_terminated_length": 17.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07266517356038094,
"epoch": 2.4609375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.009694607928395271,
"kl": 0.0022006782237440348,
"learning_rate": 4.909222113625545e-07,
"loss": 2.202560062869452e-05,
"num_tokens": 1515614.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 315,
"step_time": 6.553128839000237
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 16.25,
"completions/mean_terminated_length": 16.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1260695531964302,
"epoch": 2.46875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.02785095013678074,
"kl": 0.003913273336365819,
"learning_rate": 4.774575140626317e-07,
"loss": 3.9146947528934106e-05,
"num_tokens": 1521120.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 316,
"step_time": 7.450610239000071
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 29.0,
"completions/mean_terminated_length": 29.0,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"entropy": 0.02743313182145357,
"epoch": 2.4765625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0003154528676532209,
"kl": 0.0018573110573925078,
"learning_rate": 4.6416052570020047e-07,
"loss": 1.8573109628050588e-05,
"num_tokens": 1525348.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 317,
"step_time": 6.067128046000107
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 27.5,
"completions/mean_terminated_length": 27.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03275044448673725,
"epoch": 2.484375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0030340480152517557,
"kl": 0.0018103363690897822,
"learning_rate": 4.510323488596588e-07,
"loss": 1.801705002435483e-05,
"num_tokens": 1529396.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 318,
"step_time": 6.102241871999922
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 60.0,
"completions/max_terminated_length": 60.0,
"completions/mean_length": 31.25,
"completions/mean_terminated_length": 31.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.044731512665748596,
"epoch": 2.4921875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006392910145223141,
"kl": 0.002438001334667206,
"learning_rate": 4.380740721275786e-07,
"loss": 2.4828672394505702e-05,
"num_tokens": 1533658.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 319,
"step_time": 8.285051075999945
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 37.0,
"completions/max_terminated_length": 37.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11057645827531815,
"epoch": 2.5,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03642839938402176,
"kl": 0.0035907960700569674,
"learning_rate": 4.252867700024374e-07,
"loss": 4.713074667961337e-05,
"num_tokens": 1539236.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 320,
"step_time": 7.745400238999991
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.875,
"completions/mean_terminated_length": 19.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06698767095804214,
"epoch": 2.5078125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007277606055140495,
"kl": 0.0023683570325374603,
"learning_rate": 4.1267150280552256e-07,
"loss": 2.3156695533543825e-05,
"num_tokens": 1543991.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 321,
"step_time": 7.041696197999954
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.625,
"completions/mean_terminated_length": 13.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12220295518636703,
"epoch": 2.515625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.021708490327000618,
"kl": 0.005996785359457135,
"learning_rate": 4.002293165930088e-07,
"loss": 5.987433178233914e-05,
"num_tokens": 1549476.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 322,
"step_time": 5.8188227460000235
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 21.375,
"completions/mean_terminated_length": 21.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.060368530452251434,
"epoch": 2.5234375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03838368505239487,
"kl": 0.009895860450342298,
"learning_rate": 3.879612430692223e-07,
"loss": 8.399881335208192e-05,
"num_tokens": 1554271.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 323,
"step_time": 6.701541300000372
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.375,
"completions/mean_terminated_length": 19.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07650637812912464,
"epoch": 2.53125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004762283060699701,
"kl": 0.0014340974448714405,
"learning_rate": 3.7586829950108787e-07,
"loss": 1.502779468864901e-05,
"num_tokens": 1558994.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 324,
"step_time": 6.7548362940001425
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 66.0,
"completions/max_terminated_length": 66.0,
"completions/mean_length": 29.375,
"completions/mean_terminated_length": 29.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.2658330798149109,
"epoch": 2.5390625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.003241270314902067,
"kl": 0.0007280520221684128,
"learning_rate": 3.639514886337786e-07,
"loss": 8.224497832998168e-06,
"num_tokens": 1564653.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 325,
"step_time": 10.17069353899933
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07305469736456871,
"epoch": 2.546875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0017727442318573594,
"kl": 0.0009516599820926785,
"learning_rate": 3.5221179860757156e-07,
"loss": 9.93437697616173e-06,
"num_tokens": 1569417.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 326,
"step_time": 6.583777612000176
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 23.0,
"completions/mean_terminated_length": 23.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03469938971102238,
"epoch": 2.5546875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008155560120940208,
"kl": 0.0023509636521339417,
"learning_rate": 3.4065020287590456e-07,
"loss": 2.3022465029498562e-05,
"num_tokens": 1573341.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 327,
"step_time": 5.557240961000389
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 25.0,
"completions/mean_terminated_length": 25.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.029884813353419304,
"epoch": 2.5625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0019974284805357456,
"kl": 0.0016899254987947643,
"learning_rate": 3.292676601246661e-07,
"loss": 1.6899255570024252e-05,
"num_tokens": 1577501.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 328,
"step_time": 5.841788397000528
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 23.25,
"completions/mean_terminated_length": 23.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03349179029464722,
"epoch": 2.5703125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006807905156165361,
"kl": 0.002164472476579249,
"learning_rate": 3.18065114192693e-07,
"loss": 2.147164923371747e-05,
"num_tokens": 1581531.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 329,
"step_time": 5.71764247999954
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 20.75,
"completions/mean_terminated_length": 20.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.10336236655712128,
"epoch": 2.578125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.013856982812285423,
"kl": 0.003586582955904305,
"learning_rate": 3.0704349399351437e-07,
"loss": 3.917415233445354e-05,
"num_tokens": 1586265.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 330,
"step_time": 6.691765174000011
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 25.125,
"completions/mean_terminated_length": 25.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0626723226159811,
"epoch": 2.5859375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004872177727520466,
"kl": 0.0017109083128161728,
"learning_rate": 2.962037134383211e-07,
"loss": 1.6971382137853652e-05,
"num_tokens": 1591170.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 331,
"step_time": 7.28137495899955
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.375,
"completions/mean_terminated_length": 13.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1446835845708847,
"epoch": 2.59375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.03924147039651871,
"kl": 0.0011937321105506271,
"learning_rate": 2.855466713601868e-07,
"loss": 1.1964344594161958e-05,
"num_tokens": 1596649.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 332,
"step_time": 5.465260511999986
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 25.25,
"completions/mean_terminated_length": 25.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05687732808291912,
"epoch": 2.6015625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0024226743262261152,
"kl": 0.001208013971336186,
"learning_rate": 2.750732514395363e-07,
"loss": 1.1762923350033816e-05,
"num_tokens": 1601571.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 333,
"step_time": 7.622067566999704
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.5,
"completions/mean_terminated_length": 19.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0665333904325962,
"epoch": 2.609375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.009631386026740074,
"kl": 0.002712785266339779,
"learning_rate": 2.647843221308721e-07,
"loss": 2.844407754309941e-05,
"num_tokens": 1606451.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 334,
"step_time": 6.9531723879999845
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 77.0,
"completions/max_terminated_length": 77.0,
"completions/mean_length": 23.875,
"completions/mean_terminated_length": 23.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.371206421405077,
"epoch": 2.6171875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00424219761043787,
"kl": 0.0011196635314263403,
"learning_rate": 2.5468073659076e-07,
"loss": 1.2226804756210186e-05,
"num_tokens": 1612018.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 335,
"step_time": 10.750021371999992
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06773288920521736,
"epoch": 2.625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.001815224764868617,
"kl": 0.001531377958599478,
"learning_rate": 2.44763332607087e-07,
"loss": 1.508441346231848e-05,
"num_tokens": 1616870.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 336,
"step_time": 7.147886858999755
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 27.25,
"completions/mean_terminated_length": 27.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05348302982747555,
"epoch": 2.6328125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.002274098340421915,
"kl": 0.0012252243468537927,
"learning_rate": 2.3503293252959136e-07,
"loss": 1.2252243323018774e-05,
"num_tokens": 1621724.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 337,
"step_time": 7.523159131000284
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 39.0,
"completions/max_terminated_length": 39.0,
"completions/mean_length": 23.625,
"completions/mean_terminated_length": 23.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1177109070122242,
"epoch": 2.640625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.011636418290436268,
"kl": 0.003375065280124545,
"learning_rate": 2.2549034320167501e-07,
"loss": 3.056726563954726e-05,
"num_tokens": 1626517.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 338,
"step_time": 7.475345586000003
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07587442919611931,
"epoch": 2.6484375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.003583204001188278,
"kl": 0.001523865619674325,
"learning_rate": 2.1613635589349756e-07,
"loss": 1.486748533352511e-05,
"num_tokens": 1631389.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 339,
"step_time": 6.883027566999772
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 27.0,
"completions/mean_terminated_length": 27.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.02992851845920086,
"epoch": 2.65625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0010637313826009631,
"kl": 0.0017816239269450307,
"learning_rate": 2.0697174623636795e-07,
"loss": 1.7696058421279304e-05,
"num_tokens": 1635621.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 340,
"step_time": 5.765514154000357
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 40.0,
"completions/max_terminated_length": 40.0,
"completions/mean_length": 24.375,
"completions/mean_terminated_length": 24.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06319654919207096,
"epoch": 2.6640625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004237866494804621,
"kl": 0.0019926169188693166,
"learning_rate": 1.9799727415842323e-07,
"loss": 2.3374921511276625e-05,
"num_tokens": 1640408.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 341,
"step_time": 7.587094982999588
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 16.125,
"completions/mean_terminated_length": 16.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12324276193976402,
"epoch": 2.671875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004930234979838133,
"kl": 0.0005672876577591524,
"learning_rate": 1.8921368382162352e-07,
"loss": 6.185021447890904e-06,
"num_tokens": 1645961.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 342,
"step_time": 7.826221096000154
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 115.0,
"completions/max_terminated_length": 115.0,
"completions/mean_length": 34.25,
"completions/mean_terminated_length": 34.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0991355199366808,
"epoch": 2.6796875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0027120131999254227,
"kl": 0.0009974082058761269,
"learning_rate": 1.8062170356003854e-07,
"loss": 9.134013453149237e-06,
"num_tokens": 1650807.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 343,
"step_time": 13.32459857799995
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.875,
"completions/mean_terminated_length": 21.875,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07806593738496304,
"epoch": 2.6875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0022448606323450804,
"kl": 0.0018360121175646782,
"learning_rate": 1.7222204581946038e-07,
"loss": 1.7249569282284938e-05,
"num_tokens": 1655702.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 344,
"step_time": 6.476274790999923
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 22.0,
"completions/mean_terminated_length": 22.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.0640562754124403,
"epoch": 2.6953125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.013984336517751217,
"kl": 0.002910680166678503,
"learning_rate": 1.6401540709832242e-07,
"loss": 3.365988959558308e-05,
"num_tokens": 1660514.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 345,
"step_time": 7.252558954999586
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 82.0,
"completions/max_terminated_length": 82.0,
"completions/mean_length": 33.625,
"completions/mean_terminated_length": 33.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03660286124795675,
"epoch": 2.703125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.003199823433533311,
"kl": 0.0017720660544000566,
"learning_rate": 1.5600246788994938e-07,
"loss": 1.750032060954254e-05,
"num_tokens": 1664703.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 346,
"step_time": 9.787063931000375
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 19.625,
"completions/mean_terminated_length": 19.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06424449197947979,
"epoch": 2.7109375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0034813072998076677,
"kl": 0.0016274115769192576,
"learning_rate": 1.4818389262612948e-07,
"loss": 1.6602698451606557e-05,
"num_tokens": 1669584.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 347,
"step_time": 6.833064813999499
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 27.25,
"completions/mean_terminated_length": 27.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06542891636490822,
"epoch": 2.71875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0009135848376899958,
"kl": 0.0008284190844278783,
"learning_rate": 1.4056032962202038e-07,
"loss": 8.49508069222793e-06,
"num_tokens": 1674430.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 348,
"step_time": 7.179126660000293
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.625,
"completions/mean_terminated_length": 13.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1259615123271942,
"epoch": 2.7265625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006351651158183813,
"kl": 0.0011312126298435032,
"learning_rate": 1.3313241102239056e-07,
"loss": 1.1335807357681915e-05,
"num_tokens": 1679939.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 349,
"step_time": 5.610404259999996
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 23.0,
"completions/mean_terminated_length": 23.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.032224283553659916,
"epoch": 2.734375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0039573912508785725,
"kl": 0.0017619336140342057,
"learning_rate": 1.2590075274920206e-07,
"loss": 1.7461547031416558e-05,
"num_tokens": 1684007.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 350,
"step_time": 5.557787984000242
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.5,
"completions/mean_terminated_length": 21.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06698081642389297,
"epoch": 2.7421875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.002890112577006221,
"kl": 0.0011417069763410836,
"learning_rate": 1.1886595445053745e-07,
"loss": 1.179358150693588e-05,
"num_tokens": 1688915.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 351,
"step_time": 6.857757833000051
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 21.625,
"completions/mean_terminated_length": 21.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.05869276635348797,
"epoch": 2.75,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0013901939382776618,
"kl": 0.0012589980033226311,
"learning_rate": 1.120285994508799e-07,
"loss": 1.2590622645802796e-05,
"num_tokens": 1693788.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 352,
"step_time": 6.611652337999658
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 37.0,
"completions/max_terminated_length": 37.0,
"completions/mean_length": 20.375,
"completions/mean_terminated_length": 20.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07492958009243011,
"epoch": 2.7578125,
"frac_reward_zero_std": 0.5,
"grad_norm": 2.069866895675659,
"kl": 0.37729693634901196,
"learning_rate": 1.053892547027402e-07,
"loss": -0.09150571376085281,
"num_tokens": 1698687.0,
"reward": 0.887499988079071,
"reward_std": 0.3181980550289154,
"rewards/reward_fn/mean": 0.887499988079071,
"rewards/reward_fn/std": 0.3181980550289154,
"step": 353,
"step_time": 7.622229282999797
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.125,
"completions/mean_terminated_length": 19.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07208308577537537,
"epoch": 2.765625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.001533597824163735,
"kl": 0.0011728731915354729,
"learning_rate": 9.894847073964875e-08,
"loss": 1.1982096111751162e-05,
"num_tokens": 1703472.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 354,
"step_time": 6.501406307999787
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.375,
"completions/mean_terminated_length": 13.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13054991513490677,
"epoch": 2.7734375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.008911830373108387,
"kl": 0.0018170861876569688,
"learning_rate": 9.270678163050218e-08,
"loss": 1.823622551455628e-05,
"num_tokens": 1708979.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 355,
"step_time": 5.554297769000186
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 23.0,
"completions/mean_terminated_length": 23.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.036714909598231316,
"epoch": 2.78125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006838838569819927,
"kl": 0.002757440961431712,
"learning_rate": 8.666470493528007e-08,
"loss": 2.4345088604604825e-05,
"num_tokens": 1712979.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 356,
"step_time": 5.553534669000328
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 16.125,
"completions/mean_terminated_length": 16.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12004699558019638,
"epoch": 2.7890625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.003003190504387021,
"kl": 0.0006363969732774422,
"learning_rate": 8.082274166213016e-08,
"loss": 6.007579941069707e-06,
"num_tokens": 1718508.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 357,
"step_time": 7.342511635000392
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 27.0,
"completions/mean_terminated_length": 27.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.031998704187572,
"epoch": 2.796875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0018181306077167392,
"kl": 0.0013061033096164465,
"learning_rate": 7.518137622582189e-08,
"loss": 1.2957580111105926e-05,
"num_tokens": 1722584.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 358,
"step_time": 5.5584459810002045
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 25.0,
"completions/mean_terminated_length": 25.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03392230533063412,
"epoch": 2.8046875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005436773411929607,
"kl": 0.002172080159652978,
"learning_rate": 6.974107640758176e-08,
"loss": 2.063044303213246e-05,
"num_tokens": 1726536.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 359,
"step_time": 5.491190248999828
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 16.125,
"completions/mean_terminated_length": 16.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12706465646624565,
"epoch": 2.8125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.005568970926105976,
"kl": 0.0011207733186893165,
"learning_rate": 6.450229331630253e-08,
"loss": 1.0281852155458182e-05,
"num_tokens": 1732041.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 360,
"step_time": 7.323580845000379
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 27.0,
"completions/mean_terminated_length": 27.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.028290102258324623,
"epoch": 2.8203125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0013824108755216002,
"kl": 0.0015246181283146143,
"learning_rate": 5.946546135113862e-08,
"loss": 1.5194093066384085e-05,
"num_tokens": 1736217.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 361,
"step_time": 5.700615629000367
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 26.75,
"completions/mean_terminated_length": 26.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06074472516775131,
"epoch": 2.828125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.003173437202349305,
"kl": 0.0010228622995782644,
"learning_rate": 5.463099816548578e-08,
"loss": 1.052904008247424e-05,
"num_tokens": 1741115.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 362,
"step_time": 7.050874709999789
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 40.0,
"completions/max_terminated_length": 40.0,
"completions/mean_length": 20.375,
"completions/mean_terminated_length": 20.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07005861215293407,
"epoch": 2.8359375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0047007217071950436,
"kl": 0.001660762238316238,
"learning_rate": 4.999930463234964e-08,
"loss": 1.6024239812395535e-05,
"num_tokens": 1746014.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 363,
"step_time": 7.63879414999974
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 16.125,
"completions/mean_terminated_length": 16.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12410368397831917,
"epoch": 2.84375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.011131868697702885,
"kl": 0.0020194172975607216,
"learning_rate": 4.557076481110367e-08,
"loss": 2.2339860151987523e-05,
"num_tokens": 1751543.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 364,
"step_time": 7.354793092999898
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.125,
"completions/mean_terminated_length": 13.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.1403006836771965,
"epoch": 2.8515625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00545878428965807,
"kl": 0.0008273630810435861,
"learning_rate": 4.134574591564494e-08,
"loss": 8.271908882306889e-06,
"num_tokens": 1757048.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 365,
"step_time": 5.55601100199965
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 23.125,
"completions/mean_terminated_length": 23.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07144136913120747,
"epoch": 2.859375,
"frac_reward_zero_std": 0.5,
"grad_norm": 0.8153831958770752,
"kl": 0.6430793326580897,
"learning_rate": 3.732459828394402e-08,
"loss": -0.04161795973777771,
"num_tokens": 1761901.0,
"reward": 0.8812500238418579,
"reward_std": 0.3358757197856903,
"rewards/reward_fn/mean": 0.8812500238418579,
"rewards/reward_fn/std": 0.3358757197856903,
"step": 366,
"step_time": 7.062660026999765
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 25.125,
"completions/mean_terminated_length": 25.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06979860179126263,
"epoch": 2.8671875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004871731158345938,
"kl": 0.0016156822093762457,
"learning_rate": 3.3507655348995194e-08,
"loss": 1.5751948012621142e-05,
"num_tokens": 1766674.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 367,
"step_time": 7.081914410000536
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 19.25,
"completions/mean_terminated_length": 19.25,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.10976235195994377,
"epoch": 2.875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004110577516257763,
"kl": 0.0005710943078156561,
"learning_rate": 2.98952336111677e-08,
"loss": 6.769578249077313e-06,
"num_tokens": 1772252.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 368,
"step_time": 7.688176613999985
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 84.0,
"completions/max_terminated_length": 84.0,
"completions/mean_length": 34.625,
"completions/mean_terminated_length": 34.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.04085162188857794,
"epoch": 2.8828125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.002521694405004382,
"kl": 0.0012893079547211528,
"learning_rate": 2.6487632611962578e-08,
"loss": 1.2892131053376943e-05,
"num_tokens": 1776517.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 369,
"step_time": 9.961404026999844
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.08238299190998077,
"epoch": 2.890625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007919838652014732,
"kl": 0.002668961475137621,
"learning_rate": 2.3285134909173113e-08,
"loss": 2.5326695322291926e-05,
"num_tokens": 1781345.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 370,
"step_time": 6.4334861359998285
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 40.0,
"completions/max_terminated_length": 40.0,
"completions/mean_length": 16.5,
"completions/mean_terminated_length": 16.5,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12997304648160934,
"epoch": 2.8984375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00875998567789793,
"kl": 0.0007102387316990644,
"learning_rate": 2.028800605345771e-08,
"loss": 6.7351170400797855e-06,
"num_tokens": 1786897.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 371,
"step_time": 7.530595849999827
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07069241255521774,
"epoch": 2.90625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0031238263472914696,
"kl": 0.0017178311245515943,
"learning_rate": 1.7496494566317247e-08,
"loss": 1.785397034836933e-05,
"num_tokens": 1791725.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 372,
"step_time": 6.554121677999774
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 16.125,
"completions/mean_terminated_length": 16.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.11858610808849335,
"epoch": 2.9140625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.01251543965190649,
"kl": 0.0027800038806162775,
"learning_rate": 1.4910831919490997e-08,
"loss": 2.9193033697083592e-05,
"num_tokens": 1797254.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 373,
"step_time": 7.373935341000106
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 36.0,
"completions/max_terminated_length": 36.0,
"completions/mean_length": 16.375,
"completions/mean_terminated_length": 16.375,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12266703695058823,
"epoch": 2.921875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.04878854751586914,
"kl": 0.011298867233563215,
"learning_rate": 1.2531232515760328e-08,
"loss": 9.663843229645863e-05,
"num_tokens": 1802761.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 374,
"step_time": 7.402829858999667
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 30.0,
"completions/max_terminated_length": 30.0,
"completions/mean_length": 25.75,
"completions/mean_terminated_length": 25.75,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.03546947054564953,
"epoch": 2.9296875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006199007388204336,
"kl": 0.001919182192068547,
"learning_rate": 1.0357893671171793e-08,
"loss": 1.9191820683772676e-05,
"num_tokens": 1806727.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 375,
"step_time": 5.591665662999731
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 27.0,
"completions/mean_terminated_length": 27.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.030339499935507774,
"epoch": 2.9375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.001278455718420446,
"kl": 0.001506837084889412,
"learning_rate": 8.390995598676067e-09,
"loss": 1.4967106835683808e-05,
"num_tokens": 1810895.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 376,
"step_time": 5.731536610000148
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 81.0,
"completions/max_terminated_length": 81.0,
"completions/mean_length": 35.5,
"completions/mean_terminated_length": 35.5,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"entropy": 0.03009831439703703,
"epoch": 2.9453125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006957578472793102,
"kl": 0.001761360326781869,
"learning_rate": 6.63070139318378e-09,
"loss": 1.8286591512151062e-05,
"num_tokens": 1815119.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 377,
"step_time": 9.591905867999685
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 19.0,
"completions/mean_terminated_length": 19.0,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.07046742364764214,
"epoch": 2.953125,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.007482482586055994,
"kl": 0.0023475169437006116,
"learning_rate": 5.077157018041623e-09,
"loss": 2.212448998761829e-05,
"num_tokens": 1819847.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 378,
"step_time": 6.737704846999804
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 14.0,
"completions/mean_terminated_length": 14.0,
"completions/min_length": 14.0,
"completions/min_terminated_length": 14.0,
"entropy": 0.12937615811824799,
"epoch": 2.9609375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.023704534396529198,
"kl": 0.004825950600206852,
"learning_rate": 3.730491292930072e-09,
"loss": 4.825950600206852e-05,
"num_tokens": 1825355.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 379,
"step_time": 5.74336707700013
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 14.0,
"completions/max_terminated_length": 14.0,
"completions/mean_length": 13.125,
"completions/mean_terminated_length": 13.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.13150636106729507,
"epoch": 2.96875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.004724206868559122,
"kl": 0.0005312660941854119,
"learning_rate": 2.590815883181108e-09,
"loss": 5.310364940669388e-06,
"num_tokens": 1830860.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 380,
"step_time": 5.56844295000019
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 16.125,
"completions/mean_terminated_length": 16.125,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.12551702931523323,
"epoch": 2.9765625,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0012423633597791195,
"kl": 0.00047561304381815717,
"learning_rate": 1.6582252905186779e-09,
"loss": 5.338163646229077e-06,
"num_tokens": 1836389.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 381,
"step_time": 7.517625181999847
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 38.0,
"completions/max_terminated_length": 38.0,
"completions/mean_length": 25.625,
"completions/mean_terminated_length": 25.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06911934912204742,
"epoch": 2.984375,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.006786187645047903,
"kl": 0.0018059620633721352,
"learning_rate": 9.32796845223294e-10,
"loss": 1.838826938183047e-05,
"num_tokens": 1841294.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 382,
"step_time": 7.187780472999748
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 102.0,
"completions/max_terminated_length": 102.0,
"completions/mean_length": 28.625,
"completions/mean_terminated_length": 28.625,
"completions/min_length": 13.0,
"completions/min_terminated_length": 13.0,
"entropy": 0.06470359116792679,
"epoch": 2.9921875,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.00552311772480607,
"kl": 0.0018424552981741726,
"learning_rate": 4.1459069971938604e-10,
"loss": 1.6110756405396387e-05,
"num_tokens": 1846227.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 383,
"step_time": 12.270986474999972
},
{
"clip_ratio/high_max": 0.0,
"clip_ratio/high_mean": 0.0,
"clip_ratio/low_mean": 0.0,
"clip_ratio/low_min": 0.0,
"clip_ratio/region_mean": 0.0,
"completions/clipped_ratio": 0.0,
"completions/max_length": 29.0,
"completions/max_terminated_length": 29.0,
"completions/mean_length": 29.0,
"completions/mean_terminated_length": 29.0,
"completions/min_length": 29.0,
"completions/min_terminated_length": 29.0,
"entropy": 0.028123882599174976,
"epoch": 3.0,
"frac_reward_zero_std": 1.0,
"grad_norm": 0.0005120371351949871,
"kl": 0.0014926688745617867,
"learning_rate": 1.0364982358707087e-10,
"loss": 1.4926688891137019e-05,
"num_tokens": 1850395.0,
"reward": 1.0,
"reward_std": 0.0,
"rewards/reward_fn/mean": 1.0,
"rewards/reward_fn/std": 0.0,
"step": 384,
"step_time": 5.804545400000279
}
],
"logging_steps": 1,
"max_steps": 384,
"num_input_tokens_seen": 1850395,
"num_train_epochs": 3,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 0.0,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}