sravanthib's picture
Add files using upload-large-folder tool
c89ec76 verified
Raw
History Blame Contribute Delete
203 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 99.82758620689656,
"eval_steps": 50,
"global_step": 600,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"batch_accuracy": 0.4375,
"clip_ratio": 0.0,
"completion_length": 245.16518783569336,
"epoch": 0.13793103448275862,
"grad_norm": 9.47688737915962,
"kl": 4.5859375,
"learning_rate": 0.0,
"loss": -0.0177,
"reward": 0.4375000260770321,
"reward_std": 0.18501290306448936,
"rewards/unified_reward_func": 0.4375000260770321,
"step": 1
},
{
"clip_ratio": 0.0,
"epoch": 0.27586206896551724,
"grad_norm": 9.476790609815902,
"kl": 4.5859375,
"learning_rate": 5e-08,
"loss": -0.0177,
"step": 2
},
{
"batch_accuracy": 0.6473214285714286,
"clip_ratio": 0.0,
"completion_length": 257.0803680419922,
"epoch": 0.41379310344827586,
"grad_norm": 5.633209761896716,
"kl": 3.904296875,
"learning_rate": 1e-07,
"loss": 0.0136,
"reward": 0.6473214626312256,
"reward_std": 0.17751124873757362,
"rewards/unified_reward_func": 0.6473214626312256,
"step": 3
},
{
"clip_ratio": 0.004003038338851184,
"epoch": 0.5517241379310345,
"grad_norm": 5.597864810640234,
"kl": 3.912109375,
"learning_rate": 1.5e-07,
"loss": 0.014,
"step": 4
},
{
"batch_accuracy": 0.6651785714285714,
"clip_ratio": 0.0,
"completion_length": 262.24108123779297,
"epoch": 0.6896551724137931,
"grad_norm": 2.1149925603667534,
"kl": 1.17431640625,
"learning_rate": 2e-07,
"loss": -0.0038,
"reward": 0.6651786118745804,
"reward_std": 0.1720893383026123,
"rewards/unified_reward_func": 0.6651786118745804,
"step": 5
},
{
"clip_ratio": 0.002543629379943013,
"epoch": 0.8275862068965517,
"grad_norm": 2.0967745522222123,
"kl": 1.1083984375,
"learning_rate": 2.5e-07,
"loss": -0.004,
"step": 6
},
{
"batch_accuracy": 0.53125,
"clip_ratio": 0.0,
"completion_length": 253.30805206298828,
"epoch": 1.1379310344827587,
"grad_norm": 4.155536879447195,
"kl": 2.4609375,
"learning_rate": 3e-07,
"loss": -0.0355,
"reward": 0.5312500149011612,
"reward_std": 0.18532243371009827,
"rewards/unified_reward_func": 0.5312500149011612,
"step": 7
},
{
"clip_ratio": 0.0036200762260705233,
"epoch": 1.2758620689655173,
"grad_norm": 4.198764999207007,
"kl": 2.4296875,
"learning_rate": 3.5e-07,
"loss": -0.0353,
"step": 8
},
{
"batch_accuracy": 0.5669642857142857,
"clip_ratio": 0.0,
"completion_length": 250.44643783569336,
"epoch": 1.4137931034482758,
"grad_norm": 2.9550024325113653,
"kl": 1.6478271484375,
"learning_rate": 4e-07,
"loss": -0.0255,
"reward": 0.5669642984867096,
"reward_std": 0.14188865199685097,
"rewards/unified_reward_func": 0.5669642984867096,
"step": 9
},
{
"clip_ratio": 0.0023318197927437723,
"epoch": 1.5517241379310345,
"grad_norm": 2.9562543058271067,
"kl": 1.7484130859375,
"learning_rate": 4.5e-07,
"loss": -0.0254,
"step": 10
},
{
"batch_accuracy": 0.5491071428571428,
"clip_ratio": 0.0,
"completion_length": 282.7053756713867,
"epoch": 1.6896551724137931,
"grad_norm": 396.7503114836838,
"kl": 7.328125,
"learning_rate": 5e-07,
"loss": 0.0191,
"reward": 0.5491071715950966,
"reward_std": 0.14549032226204872,
"rewards/unified_reward_func": 0.5491071715950966,
"step": 11
},
{
"clip_ratio": 0.0031792108493391424,
"epoch": 1.8275862068965516,
"grad_norm": 956.7023024380353,
"kl": 13.0654296875,
"learning_rate": 5.5e-07,
"loss": 0.0253,
"step": 12
},
{
"batch_accuracy": 0.5133928571428571,
"clip_ratio": 0.0,
"completion_length": 308.8884048461914,
"epoch": 2.1379310344827585,
"grad_norm": 232.27657880882657,
"kl": 4.3515625,
"learning_rate": 6e-07,
"loss": -0.0097,
"reward": 0.513392873108387,
"reward_std": 0.24723730236291885,
"rewards/unified_reward_func": 0.513392873108387,
"step": 13
},
{
"clip_ratio": 0.0059004299400839955,
"epoch": 2.2758620689655173,
"grad_norm": 879.2933553212865,
"kl": 10.41015625,
"learning_rate": 6.5e-07,
"loss": -0.0028,
"step": 14
},
{
"batch_accuracy": 0.5714285714285714,
"clip_ratio": 0.0,
"completion_length": 265.87054443359375,
"epoch": 2.413793103448276,
"grad_norm": 2255.3480730668302,
"kl": 68.51708984375,
"learning_rate": 7e-07,
"loss": 0.0654,
"reward": 0.5714285969734192,
"reward_std": 0.17464973405003548,
"rewards/unified_reward_func": 0.5714285969734192,
"step": 15
},
{
"clip_ratio": 0.005311834625899792,
"epoch": 2.5517241379310347,
"grad_norm": 3452.5379008042123,
"kl": 88.62890625,
"learning_rate": 7.5e-07,
"loss": 0.0857,
"step": 16
},
{
"batch_accuracy": 0.625,
"clip_ratio": 0.0,
"completion_length": 211.33036422729492,
"epoch": 2.689655172413793,
"grad_norm": 1980.5043731662072,
"kl": 26.06103515625,
"learning_rate": 8e-07,
"loss": 0.0318,
"reward": 0.6250000223517418,
"reward_std": 0.1202336996793747,
"rewards/unified_reward_func": 0.6250000223517418,
"step": 17
},
{
"clip_ratio": 0.003107672091573477,
"epoch": 2.8275862068965516,
"grad_norm": 319.9609460610652,
"kl": 6.7763671875,
"learning_rate": 8.499999999999999e-07,
"loss": 0.014,
"step": 18
},
{
"batch_accuracy": 0.5491071428571428,
"clip_ratio": 0.0,
"completion_length": 256.5535888671875,
"epoch": 3.1379310344827585,
"grad_norm": 14.147026813832152,
"kl": 1.04296875,
"learning_rate": 9e-07,
"loss": -0.007,
"reward": 0.549107164144516,
"reward_std": 0.20905713737010956,
"rewards/unified_reward_func": 0.549107164144516,
"step": 19
},
{
"clip_ratio": 0.00605745759094134,
"epoch": 3.2758620689655173,
"grad_norm": 16.315015906851375,
"kl": 3.7421875,
"learning_rate": 9.499999999999999e-07,
"loss": -0.0025,
"step": 20
},
{
"batch_accuracy": 0.5758928571428571,
"clip_ratio": 0.0,
"completion_length": 254.38394165039062,
"epoch": 3.413793103448276,
"grad_norm": 401.95809547955486,
"kl": 27.5390625,
"learning_rate": 1e-06,
"loss": 0.0134,
"reward": 0.5758928805589676,
"reward_std": 0.1720893420279026,
"rewards/unified_reward_func": 0.5758928805589676,
"step": 21
},
{
"clip_ratio": 0.00354966675513424,
"epoch": 3.5517241379310347,
"grad_norm": 4794.277994865746,
"kl": 2.33935546875,
"learning_rate": 1e-06,
"loss": 0.348,
"step": 22
},
{
"batch_accuracy": 0.6383928571428571,
"clip_ratio": 0.0,
"completion_length": 232.91072845458984,
"epoch": 3.689655172413793,
"grad_norm": 9.457388919441831,
"kl": 5.744140625,
"learning_rate": 1e-06,
"loss": -0.0543,
"reward": 0.6383928805589676,
"reward_std": 0.20905713737010956,
"rewards/unified_reward_func": 0.6383928805589676,
"step": 23
},
{
"clip_ratio": 0.0028303218714427203,
"epoch": 3.8275862068965516,
"grad_norm": 3.7206165696876052,
"kl": 5.05078125,
"learning_rate": 1e-06,
"loss": -0.0549,
"step": 24
},
{
"batch_accuracy": 0.5491071428571429,
"clip_ratio": 0.0,
"completion_length": 266.6696548461914,
"epoch": 4.137931034482759,
"grad_norm": 3.893488451423076,
"kl": 1.9423828125,
"learning_rate": 1e-06,
"loss": -0.0473,
"reward": 0.5491071715950966,
"reward_std": 0.27956050261855125,
"rewards/unified_reward_func": 0.5491071715950966,
"step": 25
},
{
"clip_ratio": 0.0044884231756441295,
"epoch": 4.275862068965517,
"grad_norm": 4.4750395499719,
"kl": 1.837890625,
"learning_rate": 1e-06,
"loss": -0.0474,
"step": 26
},
{
"batch_accuracy": 0.5982142857142857,
"clip_ratio": 0.0,
"completion_length": 245.59375762939453,
"epoch": 4.413793103448276,
"grad_norm": 7.430927491828534,
"kl": 3.63671875,
"learning_rate": 1e-06,
"loss": 0.005,
"reward": 0.5982142984867096,
"reward_std": 0.10114361345767975,
"rewards/unified_reward_func": 0.5982142984867096,
"step": 27
},
{
"clip_ratio": 0.0019684870640048757,
"epoch": 4.551724137931035,
"grad_norm": 6.857779970270849,
"kl": 3.32568359375,
"learning_rate": 1e-06,
"loss": 0.0049,
"step": 28
},
{
"batch_accuracy": 0.5446428571428572,
"clip_ratio": 0.0,
"completion_length": 252.18750762939453,
"epoch": 4.689655172413794,
"grad_norm": 5.538508182062988,
"kl": 2.44140625,
"learning_rate": 1e-06,
"loss": -0.028,
"reward": 0.5446428805589676,
"reward_std": 0.12895501032471657,
"rewards/unified_reward_func": 0.5446428805589676,
"step": 29
},
{
"clip_ratio": 0.002361440951062832,
"epoch": 4.827586206896552,
"grad_norm": 28.554258518690066,
"kl": 1.544921875,
"learning_rate": 1e-06,
"loss": -0.0252,
"step": 30
},
{
"batch_accuracy": 0.6116071428571428,
"clip_ratio": 0.0,
"completion_length": 255.02679824829102,
"epoch": 5.137931034482759,
"grad_norm": 27.41270780548312,
"kl": 1.9775390625,
"learning_rate": 1e-06,
"loss": 0.0004,
"reward": 0.611607164144516,
"reward_std": 0.1989906169474125,
"rewards/unified_reward_func": 0.611607164144516,
"step": 31
},
{
"clip_ratio": 0.005010936001781374,
"epoch": 5.275862068965517,
"grad_norm": 351.5618938932037,
"kl": 1.869140625,
"learning_rate": 1e-06,
"loss": 0.0131,
"step": 32
},
{
"batch_accuracy": 0.5267857142857143,
"clip_ratio": 0.0,
"completion_length": 255.38840103149414,
"epoch": 5.413793103448276,
"grad_norm": 221.7307789828897,
"kl": 4.8828125,
"learning_rate": 1e-06,
"loss": -0.0152,
"reward": 0.5267857387661934,
"reward_std": 0.15872061625123024,
"rewards/unified_reward_func": 0.5267857387661934,
"step": 33
},
{
"clip_ratio": 0.003642797644715756,
"epoch": 5.551724137931035,
"grad_norm": 19.906197185838778,
"kl": 1.984375,
"learning_rate": 1e-06,
"loss": -0.0163,
"step": 34
},
{
"batch_accuracy": 0.71875,
"clip_ratio": 0.0,
"completion_length": 231.86608505249023,
"epoch": 5.689655172413794,
"grad_norm": 425.6065826516748,
"kl": 13.2099609375,
"learning_rate": 1e-06,
"loss": -0.0066,
"reward": 0.7187500447034836,
"reward_std": 0.15676921978592873,
"rewards/unified_reward_func": 0.7187500447034836,
"step": 35
},
{
"clip_ratio": 0.004165306163486093,
"epoch": 5.827586206896552,
"grad_norm": 1283.2157555121717,
"kl": 25.34765625,
"learning_rate": 1e-06,
"loss": 0.0071,
"step": 36
},
{
"batch_accuracy": 0.5625,
"clip_ratio": 0.0,
"completion_length": 243.47322463989258,
"epoch": 6.137931034482759,
"grad_norm": 476.5962581699318,
"kl": 12.490234375,
"learning_rate": 1e-06,
"loss": -0.0032,
"reward": 0.5625000298023224,
"reward_std": 0.16531942039728165,
"rewards/unified_reward_func": 0.5625000298023224,
"step": 37
},
{
"clip_ratio": 0.003403076552785933,
"epoch": 6.275862068965517,
"grad_norm": 58.50488278217316,
"kl": 4.296875,
"learning_rate": 1e-06,
"loss": -0.0074,
"step": 38
},
{
"batch_accuracy": 0.5803571428571429,
"clip_ratio": 0.0,
"completion_length": 279.6562690734863,
"epoch": 6.413793103448276,
"grad_norm": 20.57670369517026,
"kl": 1.1103515625,
"learning_rate": 1e-06,
"loss": 0.0094,
"reward": 0.580357164144516,
"reward_std": 0.18245531991124153,
"rewards/unified_reward_func": 0.580357164144516,
"step": 39
},
{
"clip_ratio": 0.003295150410849601,
"epoch": 6.551724137931035,
"grad_norm": 123.77326026058726,
"kl": 0.611328125,
"learning_rate": 1e-06,
"loss": 0.013,
"step": 40
},
{
"batch_accuracy": 0.6741071428571428,
"clip_ratio": 0.0,
"completion_length": 284.7366142272949,
"epoch": 6.689655172413794,
"grad_norm": 27.593321692714238,
"kl": 1.416015625,
"learning_rate": 1e-06,
"loss": -0.0192,
"reward": 0.6741071790456772,
"reward_std": 0.14413951337337494,
"rewards/unified_reward_func": 0.6741071790456772,
"step": 41
},
{
"clip_ratio": 0.0034483993076719344,
"epoch": 6.827586206896552,
"grad_norm": 9.09534191384267,
"kl": 1.541015625,
"learning_rate": 1e-06,
"loss": -0.0186,
"step": 42
},
{
"batch_accuracy": 0.6160714285714286,
"clip_ratio": 0.0,
"completion_length": 271.86609268188477,
"epoch": 7.137931034482759,
"grad_norm": 4.7592529588927315,
"kl": 1.0712890625,
"learning_rate": 1e-06,
"loss": -0.0101,
"reward": 0.6160714477300644,
"reward_std": 0.20185214653611183,
"rewards/unified_reward_func": 0.6160714477300644,
"step": 43
},
{
"clip_ratio": 0.003981543180998415,
"epoch": 7.275862068965517,
"grad_norm": 16.951873635923167,
"kl": 0.806640625,
"learning_rate": 1e-06,
"loss": -0.0106,
"step": 44
},
{
"batch_accuracy": 0.6383928571428572,
"clip_ratio": 0.0,
"completion_length": 228.32144165039062,
"epoch": 7.413793103448276,
"grad_norm": 8.354185239868439,
"kl": 0.94140625,
"learning_rate": 1e-06,
"loss": -0.02,
"reward": 0.6383928954601288,
"reward_std": 0.21417231857776642,
"rewards/unified_reward_func": 0.6383928954601288,
"step": 45
},
{
"clip_ratio": 0.002367985143791884,
"epoch": 7.551724137931035,
"grad_norm": 8.373422170690585,
"kl": 0.888671875,
"learning_rate": 1e-06,
"loss": -0.0204,
"step": 46
},
{
"batch_accuracy": 0.5669642857142857,
"clip_ratio": 0.0,
"completion_length": 287.12500762939453,
"epoch": 7.689655172413794,
"grad_norm": 21.451464508489934,
"kl": 0.85693359375,
"learning_rate": 1e-06,
"loss": 0.0002,
"reward": 0.5669643133878708,
"reward_std": 0.1665289979428053,
"rewards/unified_reward_func": 0.5669643133878708,
"step": 47
},
{
"clip_ratio": 0.002784289186820388,
"epoch": 7.827586206896552,
"grad_norm": 32.16378042731212,
"kl": 1.087890625,
"learning_rate": 1e-06,
"loss": 0.0011,
"step": 48
},
{
"batch_accuracy": 0.6696428571428572,
"clip_ratio": 0.0,
"completion_length": 251.9107208251953,
"epoch": 8.137931034482758,
"grad_norm": 15.579813273195894,
"kl": 0.83251953125,
"learning_rate": 1e-06,
"loss": -0.0025,
"reward": 0.6696428805589676,
"reward_std": 0.15464391559362411,
"rewards/unified_reward_func": 0.6696428805589676,
"step": 49
},
{
"epoch": 8.275862068965518,
"grad_norm": 1065.9074173474285,
"learning_rate": 1e-06,
"loss": 0.0116,
"step": 50
},
{
"epoch": 8.275862068965518,
"eval_batch_accuracy": 0.6047619047619047,
"eval_clip_ratio": 0.0,
"eval_completion_length": 270.27631123860675,
"eval_kl": 2.047005208333333,
"eval_loss": -0.0024145618081092834,
"eval_reward": 0.6047619342803955,
"eval_reward_std": 0.1999184677998225,
"eval_rewards/unified_reward_func": 0.6047619342803955,
"eval_runtime": 336.3398,
"eval_samples_per_second": 0.297,
"eval_steps_per_second": 0.006,
"step": 50
},
{
"batch_accuracy": 0.5491071428571428,
"clip_ratio": 0.001715210877591744,
"completion_length": 287.3884048461914,
"epoch": 8.413793103448276,
"grad_norm": 42.612573532085165,
"kl": 7.935546875,
"learning_rate": 1e-06,
"loss": -0.003,
"reward": 0.5491071566939354,
"reward_std": 0.15555683337152004,
"rewards/unified_reward_func": 0.5491071566939354,
"step": 51
},
{
"clip_ratio": 0.002908625523559749,
"epoch": 8.551724137931034,
"grad_norm": 89.91155792984003,
"kl": 0.63134765625,
"learning_rate": 1e-06,
"loss": -0.0032,
"step": 52
},
{
"batch_accuracy": 0.6473214285714286,
"clip_ratio": 0.0,
"completion_length": 255.51787185668945,
"epoch": 8.689655172413794,
"grad_norm": 1.3501585457962537,
"kl": 0.81201171875,
"learning_rate": 1e-06,
"loss": -0.0016,
"reward": 0.6473214775323868,
"reward_std": 0.16006861999630928,
"rewards/unified_reward_func": 0.6473214775323868,
"step": 53
},
{
"clip_ratio": 0.0019645251450128853,
"epoch": 8.827586206896552,
"grad_norm": 5.7495234417840075,
"kl": 0.70654296875,
"learning_rate": 1e-06,
"loss": -0.0028,
"step": 54
},
{
"batch_accuracy": 0.6160714285714286,
"clip_ratio": 0.0,
"completion_length": 285.558048248291,
"epoch": 9.137931034482758,
"grad_norm": 22.095081988775654,
"kl": 0.93408203125,
"learning_rate": 1e-06,
"loss": 0.0001,
"reward": 0.6160714477300644,
"reward_std": 0.16006582230329514,
"rewards/unified_reward_func": 0.6160714477300644,
"step": 55
},
{
"clip_ratio": 0.0038310182571876794,
"epoch": 9.275862068965518,
"grad_norm": 70.73324902480591,
"kl": 7.00390625,
"learning_rate": 1e-06,
"loss": 0.0073,
"step": 56
},
{
"batch_accuracy": 0.6473214285714286,
"clip_ratio": 0.0,
"completion_length": 251.6428680419922,
"epoch": 9.413793103448276,
"grad_norm": 10020.221677903297,
"kl": 114.0625,
"learning_rate": 1e-06,
"loss": 0.1067,
"reward": 0.6473214626312256,
"reward_std": 0.22935960441827774,
"rewards/unified_reward_func": 0.6473214626312256,
"step": 57
},
{
"clip_ratio": 0.005764591624028981,
"epoch": 9.551724137931034,
"grad_norm": 95.41037910173195,
"kl": 5.66015625,
"learning_rate": 1e-06,
"loss": 0.0077,
"step": 58
},
{
"batch_accuracy": 0.6651785714285714,
"clip_ratio": 0.0,
"completion_length": 242.2500114440918,
"epoch": 9.689655172413794,
"grad_norm": 264.24354788959295,
"kl": 16.71875,
"learning_rate": 1e-06,
"loss": 0.0046,
"reward": 0.6651786118745804,
"reward_std": 0.08552403189241886,
"rewards/unified_reward_func": 0.6651786118745804,
"step": 59
},
{
"clip_ratio": 0.0016965966206043959,
"epoch": 9.827586206896552,
"grad_norm": 41.8877627907783,
"kl": 17.09375,
"learning_rate": 1e-06,
"loss": 0.005,
"step": 60
},
{
"batch_accuracy": 0.5580357142857143,
"clip_ratio": 0.0,
"completion_length": 287.27233505249023,
"epoch": 10.137931034482758,
"grad_norm": 9.649367519026018,
"kl": 4.73828125,
"learning_rate": 1e-06,
"loss": 0.0293,
"reward": 0.5580357313156128,
"reward_std": 0.18202302604913712,
"rewards/unified_reward_func": 0.5580357313156128,
"step": 61
},
{
"clip_ratio": 0.003987273928942159,
"epoch": 10.275862068965518,
"grad_norm": 4.26991643094351,
"kl": 1.8408203125,
"learning_rate": 1e-06,
"loss": 0.0287,
"step": 62
},
{
"batch_accuracy": 0.6875,
"clip_ratio": 0.0,
"completion_length": 247.79465103149414,
"epoch": 10.413793103448276,
"grad_norm": 5.240743579266302,
"kl": 2.392578125,
"learning_rate": 1e-06,
"loss": 0.0199,
"reward": 0.6875000149011612,
"reward_std": 0.165928415954113,
"rewards/unified_reward_func": 0.6875000149011612,
"step": 63
},
{
"clip_ratio": 0.002536454499932006,
"epoch": 10.551724137931034,
"grad_norm": 0.6982187039233699,
"kl": 1.521484375,
"learning_rate": 1e-06,
"loss": 0.0191,
"step": 64
},
{
"batch_accuracy": 0.6875,
"clip_ratio": 0.0,
"completion_length": 264.5446548461914,
"epoch": 10.689655172413794,
"grad_norm": 1.4881013556571023,
"kl": 1.2353515625,
"learning_rate": 1e-06,
"loss": -0.0058,
"reward": 0.6875000298023224,
"reward_std": 0.1382853128015995,
"rewards/unified_reward_func": 0.6875000298023224,
"step": 65
},
{
"clip_ratio": 0.003119972941931337,
"epoch": 10.827586206896552,
"grad_norm": 3.092593425215386,
"kl": 0.789794921875,
"learning_rate": 1e-06,
"loss": -0.0066,
"step": 66
},
{
"batch_accuracy": 0.6160714285714286,
"clip_ratio": 0.0,
"completion_length": 311.433048248291,
"epoch": 11.137931034482758,
"grad_norm": 1.01948057937195,
"kl": 1.1279296875,
"learning_rate": 1e-06,
"loss": 0.0049,
"reward": 0.6160714477300644,
"reward_std": 0.14548752084374428,
"rewards/unified_reward_func": 0.6160714477300644,
"step": 67
},
{
"clip_ratio": 0.002209829064668156,
"epoch": 11.275862068965518,
"grad_norm": 0.8281537495597667,
"kl": 1.1767578125,
"learning_rate": 1e-06,
"loss": 0.0042,
"step": 68
},
{
"batch_accuracy": 0.7053571428571428,
"clip_ratio": 0.0,
"completion_length": 250.76340103149414,
"epoch": 11.413793103448276,
"grad_norm": 0.9634703459739232,
"kl": 0.62939453125,
"learning_rate": 1e-06,
"loss": 0.0402,
"reward": 0.705357164144516,
"reward_std": 0.16683853790163994,
"rewards/unified_reward_func": 0.705357164144516,
"step": 69
},
{
"clip_ratio": 0.0029377865139395,
"epoch": 11.551724137931034,
"grad_norm": 1.5188872952272299,
"kl": 0.5576171875,
"learning_rate": 1e-06,
"loss": 0.0387,
"step": 70
},
{
"batch_accuracy": 0.65625,
"clip_ratio": 0.0,
"completion_length": 268.31697845458984,
"epoch": 11.689655172413794,
"grad_norm": 0.9021269984465355,
"kl": 0.8798828125,
"learning_rate": 1e-06,
"loss": -0.0067,
"reward": 0.6562500298023224,
"reward_std": 0.14097853749990463,
"rewards/unified_reward_func": 0.6562500298023224,
"step": 71
},
{
"clip_ratio": 0.0036392371403053403,
"epoch": 11.827586206896552,
"grad_norm": 0.9695227093029657,
"kl": 1.078125,
"learning_rate": 1e-06,
"loss": -0.0074,
"step": 72
},
{
"batch_accuracy": 0.5892857142857143,
"clip_ratio": 0.0,
"completion_length": 246.7187614440918,
"epoch": 12.137931034482758,
"grad_norm": 7.117785992825863,
"kl": 3.7001953125,
"learning_rate": 1e-06,
"loss": -0.0122,
"reward": 0.589285746216774,
"reward_std": 0.22229304164648056,
"rewards/unified_reward_func": 0.589285746216774,
"step": 73
},
{
"clip_ratio": 0.006052218785043806,
"epoch": 12.275862068965518,
"grad_norm": 58.99563589475248,
"kl": 1.0908203125,
"learning_rate": 1e-06,
"loss": -0.0004,
"step": 74
},
{
"batch_accuracy": 0.7723214285714286,
"clip_ratio": 0.0,
"completion_length": 265.3973274230957,
"epoch": 12.413793103448276,
"grad_norm": 1.7066461907389874,
"kl": 0.826904296875,
"learning_rate": 1e-06,
"loss": -0.004,
"reward": 0.7723214626312256,
"reward_std": 0.1415819153189659,
"rewards/unified_reward_func": 0.7723214626312256,
"step": 75
},
{
"clip_ratio": 0.002809168363455683,
"epoch": 12.551724137931034,
"grad_norm": 1.0796312368974332,
"kl": 0.443359375,
"learning_rate": 1e-06,
"loss": -0.0046,
"step": 76
},
{
"batch_accuracy": 0.6294642857142857,
"clip_ratio": 0.0,
"completion_length": 276.7232246398926,
"epoch": 12.689655172413794,
"grad_norm": 1.1083371008590541,
"kl": 0.900390625,
"learning_rate": 1e-06,
"loss": -0.0025,
"reward": 0.6294643133878708,
"reward_std": 0.14353612437844276,
"rewards/unified_reward_func": 0.6294643133878708,
"step": 77
},
{
"clip_ratio": 0.0021633205469697714,
"epoch": 12.827586206896552,
"grad_norm": 1.104357377092878,
"kl": 0.7685546875,
"learning_rate": 1e-06,
"loss": -0.0032,
"step": 78
},
{
"batch_accuracy": 0.7857142857142857,
"clip_ratio": 0.0,
"completion_length": 230.9866180419922,
"epoch": 13.137931034482758,
"grad_norm": 5.1546027891456845,
"kl": 0.912353515625,
"learning_rate": 1e-06,
"loss": 0.02,
"reward": 0.785714328289032,
"reward_std": 0.08942962624132633,
"rewards/unified_reward_func": 0.785714328289032,
"step": 79
},
{
"clip_ratio": 0.0016282629658235237,
"epoch": 13.275862068965518,
"grad_norm": 7.153630585121116,
"kl": 0.74951171875,
"learning_rate": 1e-06,
"loss": 0.0197,
"step": 80
},
{
"batch_accuracy": 0.4553571428571429,
"clip_ratio": 0.0,
"completion_length": 332.6651916503906,
"epoch": 13.413793103448276,
"grad_norm": 3.7470240632805467,
"kl": 2.3740234375,
"learning_rate": 1e-06,
"loss": -0.0099,
"reward": 0.4553571678698063,
"reward_std": 0.2024611309170723,
"rewards/unified_reward_func": 0.4553571678698063,
"step": 81
},
{
"clip_ratio": 0.003834200673736632,
"epoch": 13.551724137931034,
"grad_norm": 1.0885699596296923,
"kl": 1.330078125,
"learning_rate": 1e-06,
"loss": -0.0112,
"step": 82
},
{
"batch_accuracy": 0.7767857142857143,
"clip_ratio": 0.0,
"completion_length": 296.401798248291,
"epoch": 13.689655172413794,
"grad_norm": 1.094797967613683,
"kl": 0.8017578125,
"learning_rate": 1e-06,
"loss": 0.0183,
"reward": 0.776785746216774,
"reward_std": 0.1814168505370617,
"rewards/unified_reward_func": 0.776785746216774,
"step": 83
},
{
"clip_ratio": 0.0045213785488158464,
"epoch": 13.827586206896552,
"grad_norm": 0.8596258608296502,
"kl": 0.736328125,
"learning_rate": 1e-06,
"loss": 0.0173,
"step": 84
},
{
"batch_accuracy": 0.6696428571428571,
"clip_ratio": 0.0,
"completion_length": 266.92858123779297,
"epoch": 14.137931034482758,
"grad_norm": 0.6150343367414953,
"kl": 0.7158203125,
"learning_rate": 1e-06,
"loss": 0.0125,
"reward": 0.6696428805589676,
"reward_std": 0.07875411957502365,
"rewards/unified_reward_func": 0.6696428805589676,
"step": 85
},
{
"clip_ratio": 0.0018555940187070519,
"epoch": 14.275862068965518,
"grad_norm": 0.5097068454390808,
"kl": 0.6337890625,
"learning_rate": 1e-06,
"loss": 0.0117,
"step": 86
},
{
"batch_accuracy": 0.6919642857142857,
"clip_ratio": 0.0,
"completion_length": 254.7812614440918,
"epoch": 14.413793103448276,
"grad_norm": 7.244582112989614,
"kl": 3.82568359375,
"learning_rate": 1e-06,
"loss": 0.0225,
"reward": 0.6919643133878708,
"reward_std": 0.16336803138256073,
"rewards/unified_reward_func": 0.6919643133878708,
"step": 87
},
{
"clip_ratio": 0.0048985659959726036,
"epoch": 14.551724137931034,
"grad_norm": 18.06983316871383,
"kl": 1.03125,
"learning_rate": 1e-06,
"loss": 0.0259,
"step": 88
},
{
"batch_accuracy": 0.6964285714285714,
"clip_ratio": 0.0,
"completion_length": 294.2321586608887,
"epoch": 14.689655172413794,
"grad_norm": 1.6611859912376654,
"kl": 2.0693359375,
"learning_rate": 1e-06,
"loss": 0.0579,
"reward": 0.6964285969734192,
"reward_std": 0.16037255339324474,
"rewards/unified_reward_func": 0.6964285969734192,
"step": 89
},
{
"clip_ratio": 0.00386410066857934,
"epoch": 14.827586206896552,
"grad_norm": 1.1461095674554305,
"kl": 1.39453125,
"learning_rate": 1e-06,
"loss": 0.0566,
"step": 90
},
{
"batch_accuracy": 0.7544642857142857,
"clip_ratio": 0.0,
"completion_length": 263.4509048461914,
"epoch": 15.137931034482758,
"grad_norm": 2.7879946254038073,
"kl": 4.13623046875,
"learning_rate": 1e-06,
"loss": 0.0507,
"reward": 0.754464328289032,
"reward_std": 0.13512154668569565,
"rewards/unified_reward_func": 0.754464328289032,
"step": 91
},
{
"clip_ratio": 0.002846388262696564,
"epoch": 15.275862068965518,
"grad_norm": 0.8191932494227651,
"kl": 1.25732421875,
"learning_rate": 1e-06,
"loss": 0.0476,
"step": 92
},
{
"batch_accuracy": 0.7142857142857143,
"clip_ratio": 0.0,
"completion_length": 310.2143020629883,
"epoch": 15.413793103448276,
"grad_norm": 1.1634079207613266,
"kl": 2.3837890625,
"learning_rate": 1e-06,
"loss": 0.0156,
"reward": 0.714285746216774,
"reward_std": 0.16006582602858543,
"rewards/unified_reward_func": 0.714285746216774,
"step": 93
},
{
"clip_ratio": 0.0025695697695482522,
"epoch": 15.551724137931034,
"grad_norm": 0.9306916071768522,
"kl": 1.99169921875,
"learning_rate": 1e-06,
"loss": 0.0145,
"step": 94
},
{
"batch_accuracy": 0.6830357142857143,
"clip_ratio": 0.0,
"completion_length": 261.5401916503906,
"epoch": 15.689655172413794,
"grad_norm": 1.2265984060410038,
"kl": 1.3271484375,
"learning_rate": 1e-06,
"loss": 0.0099,
"reward": 0.6830357313156128,
"reward_std": 0.08222462423145771,
"rewards/unified_reward_func": 0.6830357313156128,
"step": 95
},
{
"clip_ratio": 0.0021528883662540466,
"epoch": 15.827586206896552,
"grad_norm": 0.7608892170172238,
"kl": 0.8759765625,
"learning_rate": 1e-06,
"loss": 0.009,
"step": 96
},
{
"batch_accuracy": 0.7276785714285714,
"clip_ratio": 0.0,
"completion_length": 297.6026916503906,
"epoch": 16.137931034482758,
"grad_norm": 0.9700441408074749,
"kl": 1.2021484375,
"learning_rate": 1e-06,
"loss": 0.0472,
"reward": 0.7276786118745804,
"reward_std": 0.21192146465182304,
"rewards/unified_reward_func": 0.7276786118745804,
"step": 97
},
{
"clip_ratio": 0.005118858069181442,
"epoch": 16.275862068965516,
"grad_norm": 1.6631282888053047,
"kl": 1.23828125,
"learning_rate": 1e-06,
"loss": 0.0458,
"step": 98
},
{
"batch_accuracy": 0.75,
"clip_ratio": 0.0,
"completion_length": 274.2142906188965,
"epoch": 16.413793103448278,
"grad_norm": 0.4317146888702674,
"kl": 1.02587890625,
"learning_rate": 1e-06,
"loss": 0.0041,
"reward": 0.7500000298023224,
"reward_std": 0.09046810120344162,
"rewards/unified_reward_func": 0.7500000298023224,
"step": 99
},
{
"epoch": 16.551724137931036,
"grad_norm": 1.133590964829475,
"learning_rate": 1e-06,
"loss": 0.0034,
"step": 100
},
{
"epoch": 16.551724137931036,
"eval_batch_accuracy": 0.6797619047619048,
"eval_clip_ratio": 0.0,
"eval_completion_length": 292.35174153645835,
"eval_kl": 0.558984375,
"eval_loss": -0.0012742819963023067,
"eval_reward": 0.6797619382540385,
"eval_reward_std": 0.1622819994886716,
"eval_rewards/unified_reward_func": 0.6797619382540385,
"eval_runtime": 357.5239,
"eval_samples_per_second": 0.28,
"eval_steps_per_second": 0.006,
"step": 100
},
{
"batch_accuracy": 0.6428571428571429,
"clip_ratio": 0.000845697577460669,
"completion_length": 257.42858123779297,
"epoch": 16.689655172413794,
"grad_norm": 1.2995812903101538,
"kl": 0.662841796875,
"learning_rate": 1e-06,
"loss": 0.018,
"reward": 0.6428571790456772,
"reward_std": 0.12054043263196945,
"rewards/unified_reward_func": 0.6428571790456772,
"step": 101
},
{
"clip_ratio": 0.0017673625843599439,
"epoch": 16.82758620689655,
"grad_norm": 0.6537204602789219,
"kl": 0.349853515625,
"learning_rate": 1e-06,
"loss": 0.0174,
"step": 102
},
{
"batch_accuracy": 0.7276785714285714,
"clip_ratio": 0.0,
"completion_length": 296.745548248291,
"epoch": 17.137931034482758,
"grad_norm": 0.9855908928682507,
"kl": 0.4176025390625,
"learning_rate": 1e-06,
"loss": 0.0449,
"reward": 0.7276785969734192,
"reward_std": 0.15555684082210064,
"rewards/unified_reward_func": 0.7276785969734192,
"step": 103
},
{
"clip_ratio": 0.004551950143650174,
"epoch": 17.275862068965516,
"grad_norm": 5.015002215768227,
"kl": 0.46875,
"learning_rate": 1e-06,
"loss": 0.0437,
"step": 104
},
{
"batch_accuracy": 0.7455357142857143,
"clip_ratio": 0.0,
"completion_length": 269.9464416503906,
"epoch": 17.413793103448278,
"grad_norm": 0.5358380953757037,
"kl": 0.333984375,
"learning_rate": 1e-06,
"loss": 0.0016,
"reward": 0.7455357611179352,
"reward_std": 0.08265971392393112,
"rewards/unified_reward_func": 0.7455357611179352,
"step": 105
},
{
"clip_ratio": 0.001166727059171535,
"epoch": 17.551724137931036,
"grad_norm": 0.6830961806380172,
"kl": 0.297119140625,
"learning_rate": 1e-06,
"loss": 0.0008,
"step": 106
},
{
"batch_accuracy": 0.7142857142857143,
"clip_ratio": 0.0,
"completion_length": 306.00001525878906,
"epoch": 17.689655172413794,
"grad_norm": 0.6823238907681224,
"kl": 0.356689453125,
"learning_rate": 1e-06,
"loss": 0.0279,
"reward": 0.7142857313156128,
"reward_std": 0.1010152529925108,
"rewards/unified_reward_func": 0.7142857313156128,
"step": 107
},
{
"clip_ratio": 0.002543082577176392,
"epoch": 17.82758620689655,
"grad_norm": 0.6313266960191072,
"kl": 0.359619140625,
"learning_rate": 1e-06,
"loss": 0.0269,
"step": 108
},
{
"batch_accuracy": 0.7321428571428571,
"clip_ratio": 0.0,
"completion_length": 287.0044746398926,
"epoch": 18.137931034482758,
"grad_norm": 0.8019490958341388,
"kl": 0.649169921875,
"learning_rate": 1e-06,
"loss": 0.0058,
"reward": 0.7321428805589676,
"reward_std": 0.13902713917195797,
"rewards/unified_reward_func": 0.7321428805589676,
"step": 109
},
{
"clip_ratio": 0.002602686858153902,
"epoch": 18.275862068965516,
"grad_norm": 0.6107705837722435,
"kl": 0.675537109375,
"learning_rate": 1e-06,
"loss": 0.0047,
"step": 110
},
{
"batch_accuracy": 0.7678571428571429,
"clip_ratio": 0.0,
"completion_length": 279.4821548461914,
"epoch": 18.413793103448278,
"grad_norm": 0.6770932627526579,
"kl": 0.353515625,
"learning_rate": 1e-06,
"loss": -0.0067,
"reward": 0.7678571790456772,
"reward_std": 0.10851971805095673,
"rewards/unified_reward_func": 0.7678571790456772,
"step": 111
},
{
"clip_ratio": 0.0023510658502345905,
"epoch": 18.551724137931036,
"grad_norm": 0.7583576047177788,
"kl": 0.293701171875,
"learning_rate": 1e-06,
"loss": -0.0072,
"step": 112
},
{
"batch_accuracy": 0.75,
"clip_ratio": 0.0,
"completion_length": 287.3214416503906,
"epoch": 18.689655172413794,
"grad_norm": 0.5279948952585755,
"kl": 0.51806640625,
"learning_rate": 1e-06,
"loss": 0.0003,
"reward": 0.7500000447034836,
"reward_std": 0.07875411584973335,
"rewards/unified_reward_func": 0.7500000447034836,
"step": 113
},
{
"clip_ratio": 0.0013883734354749322,
"epoch": 18.82758620689655,
"grad_norm": 0.38877289023644157,
"kl": 0.470703125,
"learning_rate": 1e-06,
"loss": -0.0002,
"step": 114
},
{
"batch_accuracy": 0.71875,
"clip_ratio": 0.0,
"completion_length": 271.07590103149414,
"epoch": 19.137931034482758,
"grad_norm": 0.7346182063865974,
"kl": 0.3056640625,
"learning_rate": 1e-06,
"loss": 0.0083,
"reward": 0.7187500447034836,
"reward_std": 0.12956120818853378,
"rewards/unified_reward_func": 0.7187500447034836,
"step": 115
},
{
"clip_ratio": 0.003321638738270849,
"epoch": 19.275862068965516,
"grad_norm": 0.7155226380704629,
"kl": 0.312255859375,
"learning_rate": 1e-06,
"loss": 0.0074,
"step": 116
},
{
"batch_accuracy": 0.7901785714285714,
"clip_ratio": 0.0,
"completion_length": 282.6250114440918,
"epoch": 19.413793103448278,
"grad_norm": 0.764157190415015,
"kl": 0.69677734375,
"learning_rate": 1e-06,
"loss": 0.0284,
"reward": 0.7901785969734192,
"reward_std": 0.07350331731140614,
"rewards/unified_reward_func": 0.7901785969734192,
"step": 117
},
{
"clip_ratio": 0.002791708306176588,
"epoch": 19.551724137931036,
"grad_norm": 0.9985727394911575,
"kl": 0.6328125,
"learning_rate": 1e-06,
"loss": 0.0277,
"step": 118
},
{
"batch_accuracy": 0.8392857142857143,
"clip_ratio": 0.0,
"completion_length": 250.7812614440918,
"epoch": 19.689655172413794,
"grad_norm": 0.6371173760577464,
"kl": 0.450439453125,
"learning_rate": 1e-06,
"loss": -0.001,
"reward": 0.839285746216774,
"reward_std": 0.0835726372897625,
"rewards/unified_reward_func": 0.839285746216774,
"step": 119
},
{
"clip_ratio": 0.0017081814585253596,
"epoch": 19.82758620689655,
"grad_norm": 0.49119996087886486,
"kl": 0.41259765625,
"learning_rate": 1e-06,
"loss": -0.0016,
"step": 120
},
{
"batch_accuracy": 0.8392857142857143,
"clip_ratio": 0.0,
"completion_length": 265.5759048461914,
"epoch": 20.137931034482758,
"grad_norm": 0.8976181584128184,
"kl": 0.83349609375,
"learning_rate": 1e-06,
"loss": 0.0186,
"reward": 0.839285746216774,
"reward_std": 0.0754547119140625,
"rewards/unified_reward_func": 0.839285746216774,
"step": 121
},
{
"clip_ratio": 0.0017390022403560579,
"epoch": 20.275862068965516,
"grad_norm": 0.5861725820573414,
"kl": 0.5615234375,
"learning_rate": 1e-06,
"loss": 0.0178,
"step": 122
},
{
"batch_accuracy": 0.7857142857142857,
"clip_ratio": 0.0,
"completion_length": 263.714298248291,
"epoch": 20.413793103448278,
"grad_norm": 0.46434375354901414,
"kl": 0.4921875,
"learning_rate": 1e-06,
"loss": 0.0122,
"reward": 0.7857143133878708,
"reward_std": 0.05215509608387947,
"rewards/unified_reward_func": 0.7857143133878708,
"step": 123
},
{
"clip_ratio": 0.0006422129372367635,
"epoch": 20.551724137931036,
"grad_norm": 0.4042130771183457,
"kl": 0.53564453125,
"learning_rate": 1e-06,
"loss": 0.0115,
"step": 124
},
{
"batch_accuracy": 0.75,
"clip_ratio": 0.0,
"completion_length": 243.55358123779297,
"epoch": 20.689655172413794,
"grad_norm": 0.4493382275129834,
"kl": 0.3251953125,
"learning_rate": 1e-06,
"loss": 0.0095,
"reward": 0.7500000447034836,
"reward_std": 0.04764330945909023,
"rewards/unified_reward_func": 0.7500000447034836,
"step": 125
},
{
"clip_ratio": 0.0014584372693207115,
"epoch": 20.82758620689655,
"grad_norm": 0.354450124466521,
"kl": 0.35595703125,
"learning_rate": 1e-06,
"loss": 0.009,
"step": 126
},
{
"batch_accuracy": 0.7589285714285714,
"clip_ratio": 0.0,
"completion_length": 261.5982208251953,
"epoch": 21.137931034482758,
"grad_norm": 1.1212537411147512,
"kl": 0.7021484375,
"learning_rate": 1e-06,
"loss": 0.0226,
"reward": 0.7589285969734192,
"reward_std": 0.08070831745862961,
"rewards/unified_reward_func": 0.7589285969734192,
"step": 127
},
{
"clip_ratio": 0.0021403833816293627,
"epoch": 21.275862068965516,
"grad_norm": 0.564444364760896,
"kl": 0.61328125,
"learning_rate": 1e-06,
"loss": 0.0222,
"step": 128
},
{
"batch_accuracy": 0.7991071428571428,
"clip_ratio": 0.0,
"completion_length": 283.0089416503906,
"epoch": 21.413793103448278,
"grad_norm": 1.1428331052805576,
"kl": 1.09521484375,
"learning_rate": 1e-06,
"loss": 0.0133,
"reward": 0.7991071790456772,
"reward_std": 0.0689915269613266,
"rewards/unified_reward_func": 0.7991071790456772,
"step": 129
},
{
"clip_ratio": 0.0020886963757220656,
"epoch": 21.551724137931036,
"grad_norm": 1.2791510833897302,
"kl": 0.6513671875,
"learning_rate": 1e-06,
"loss": 0.0122,
"step": 130
},
{
"batch_accuracy": 0.7544642857142857,
"clip_ratio": 0.0,
"completion_length": 264.71876525878906,
"epoch": 21.689655172413794,
"grad_norm": 0.5068843628644464,
"kl": 0.722900390625,
"learning_rate": 1e-06,
"loss": -0.0074,
"reward": 0.7544643133878708,
"reward_std": 0.04824949987232685,
"rewards/unified_reward_func": 0.7544643133878708,
"step": 131
},
{
"clip_ratio": 0.0009302312828367576,
"epoch": 21.82758620689655,
"grad_norm": 0.7591299641106868,
"kl": 0.5947265625,
"learning_rate": 1e-06,
"loss": -0.008,
"step": 132
},
{
"batch_accuracy": 0.7455357142857143,
"clip_ratio": 0.0,
"completion_length": 263.6696548461914,
"epoch": 22.137931034482758,
"grad_norm": 0.9406794669122327,
"kl": 0.609375,
"learning_rate": 1e-06,
"loss": 0.0136,
"reward": 0.745535746216774,
"reward_std": 0.12249183282256126,
"rewards/unified_reward_func": 0.745535746216774,
"step": 133
},
{
"clip_ratio": 0.0022654080821666867,
"epoch": 22.275862068965516,
"grad_norm": 4.442348454177717,
"kl": 0.37255859375,
"learning_rate": 1e-06,
"loss": 0.0144,
"step": 134
},
{
"batch_accuracy": 0.8571428571428571,
"clip_ratio": 0.0,
"completion_length": 261.57590103149414,
"epoch": 22.413793103448278,
"grad_norm": 1.2706700244569715,
"kl": 0.4873046875,
"learning_rate": 1e-06,
"loss": 0.001,
"reward": 0.8571428805589676,
"reward_std": 0.08942962251603603,
"rewards/unified_reward_func": 0.8571428805589676,
"step": 135
},
{
"clip_ratio": 0.0022475110308732837,
"epoch": 22.551724137931036,
"grad_norm": 1.758254727608865,
"kl": 0.4248046875,
"learning_rate": 1e-06,
"loss": 0.0007,
"step": 136
},
{
"batch_accuracy": 0.7991071428571428,
"clip_ratio": 0.0,
"completion_length": 257.214298248291,
"epoch": 22.689655172413794,
"grad_norm": 1.9727070210769588,
"kl": 0.457763671875,
"learning_rate": 1e-06,
"loss": 0.0432,
"reward": 0.7991071790456772,
"reward_std": 0.09619954042136669,
"rewards/unified_reward_func": 0.7991071790456772,
"step": 137
},
{
"clip_ratio": 0.0030386159487534314,
"epoch": 22.82758620689655,
"grad_norm": 2.897713421787608,
"kl": 0.53955078125,
"learning_rate": 1e-06,
"loss": 0.0428,
"step": 138
},
{
"batch_accuracy": 0.7678571428571428,
"clip_ratio": 0.0,
"completion_length": 293.4107322692871,
"epoch": 23.137931034482758,
"grad_norm": 3.4310510361220166,
"kl": 1.0546875,
"learning_rate": 1e-06,
"loss": 0.0137,
"reward": 0.7678571864962578,
"reward_std": 0.08868780359625816,
"rewards/unified_reward_func": 0.7678571864962578,
"step": 139
},
{
"clip_ratio": 0.0015242814115481451,
"epoch": 23.275862068965516,
"grad_norm": 1.4261956111254026,
"kl": 0.823486328125,
"learning_rate": 1e-06,
"loss": 0.0134,
"step": 140
},
{
"batch_accuracy": 0.7678571428571428,
"clip_ratio": 0.0,
"completion_length": 301.7500190734863,
"epoch": 23.413793103448278,
"grad_norm": 1.6514869046993512,
"kl": 0.97900390625,
"learning_rate": 1e-06,
"loss": 0.019,
"reward": 0.7678571790456772,
"reward_std": 0.08131169900298119,
"rewards/unified_reward_func": 0.7678571790456772,
"step": 141
},
{
"clip_ratio": 0.001672004524152726,
"epoch": 23.551724137931036,
"grad_norm": 0.5079581036195917,
"kl": 0.87353515625,
"learning_rate": 1e-06,
"loss": 0.0186,
"step": 142
},
{
"batch_accuracy": 0.84375,
"clip_ratio": 0.0,
"completion_length": 219.2678680419922,
"epoch": 23.689655172413794,
"grad_norm": 0.6205850478806725,
"kl": 0.560546875,
"learning_rate": 1e-06,
"loss": 0.0163,
"reward": 0.8437500447034836,
"reward_std": 0.08131450787186623,
"rewards/unified_reward_func": 0.8437500447034836,
"step": 143
},
{
"clip_ratio": 0.00177483752486296,
"epoch": 23.82758620689655,
"grad_norm": 0.6514179664692717,
"kl": 0.42724609375,
"learning_rate": 1e-06,
"loss": 0.0156,
"step": 144
},
{
"batch_accuracy": 0.7544642857142857,
"clip_ratio": 0.0,
"completion_length": 287.60269927978516,
"epoch": 24.137931034482758,
"grad_norm": 0.4866499958405398,
"kl": 0.7421875,
"learning_rate": 1e-06,
"loss": 0.03,
"reward": 0.754464328289032,
"reward_std": 0.0673395898193121,
"rewards/unified_reward_func": 0.754464328289032,
"step": 145
},
{
"clip_ratio": 0.0009519409650238231,
"epoch": 24.275862068965516,
"grad_norm": 0.36625944324763465,
"kl": 0.69873046875,
"learning_rate": 1e-06,
"loss": 0.0294,
"step": 146
},
{
"batch_accuracy": 0.7678571428571428,
"clip_ratio": 0.0,
"completion_length": 256.92412185668945,
"epoch": 24.413793103448278,
"grad_norm": 0.9361673166579322,
"kl": 0.89892578125,
"learning_rate": 1e-06,
"loss": 0.0109,
"reward": 0.7678571790456772,
"reward_std": 0.08326590247452259,
"rewards/unified_reward_func": 0.7678571790456772,
"step": 147
},
{
"clip_ratio": 0.0012340883258730173,
"epoch": 24.551724137931036,
"grad_norm": 44.342229710309866,
"kl": 1.05078125,
"learning_rate": 1e-06,
"loss": 0.0233,
"step": 148
},
{
"batch_accuracy": 0.8571428571428571,
"clip_ratio": 0.0,
"completion_length": 260.4464340209961,
"epoch": 24.689655172413794,
"grad_norm": 32.73262062897448,
"kl": 23.962890625,
"learning_rate": 1e-06,
"loss": 0.0072,
"reward": 0.85714291036129,
"reward_std": 0.09198721311986446,
"rewards/unified_reward_func": 0.85714291036129,
"step": 149
},
{
"epoch": 24.82758620689655,
"grad_norm": 1.2688282072036752,
"learning_rate": 1e-06,
"loss": -0.014,
"step": 150
},
{
"epoch": 24.82758620689655,
"eval_batch_accuracy": 0.7154761904761905,
"eval_clip_ratio": 0.0,
"eval_completion_length": 273.66440734863284,
"eval_kl": 1.7647135416666666,
"eval_loss": 0.011888084933161736,
"eval_reward": 0.7154762188593546,
"eval_reward_std": 0.11879824002583822,
"eval_rewards/unified_reward_func": 0.7154762188593546,
"eval_runtime": 345.8717,
"eval_samples_per_second": 0.289,
"eval_steps_per_second": 0.006,
"step": 150
},
{
"batch_accuracy": 0.8571428571428571,
"clip_ratio": 0.0010695938399294391,
"completion_length": 236.62500762939453,
"epoch": 25.137931034482758,
"grad_norm": 153.65426313867775,
"kl": 13.923583984375,
"learning_rate": 1e-06,
"loss": 0.0248,
"reward": 0.8571428954601288,
"reward_std": 0.08613021858036518,
"rewards/unified_reward_func": 0.8571428954601288,
"step": 151
},
{
"clip_ratio": 0.0018477260309737176,
"epoch": 25.275862068965516,
"grad_norm": 23040.052568690135,
"kl": 0.59716796875,
"learning_rate": 1e-06,
"loss": 7.2765,
"step": 152
},
{
"batch_accuracy": 0.7366071428571428,
"clip_ratio": 0.0,
"completion_length": 263.20983123779297,
"epoch": 25.413793103448278,
"grad_norm": 0.4116680840750752,
"kl": 0.911865234375,
"learning_rate": 1e-06,
"loss": 0.0034,
"reward": 0.7366071790456772,
"reward_std": 0.05441322550177574,
"rewards/unified_reward_func": 0.7366071790456772,
"step": 153
},
{
"clip_ratio": 0.0011999079579254612,
"epoch": 25.551724137931036,
"grad_norm": 0.3576925903614138,
"kl": 0.780517578125,
"learning_rate": 1e-06,
"loss": 0.0028,
"step": 154
},
{
"batch_accuracy": 0.7723214285714286,
"clip_ratio": 0.0,
"completion_length": 290.7901954650879,
"epoch": 25.689655172413794,
"grad_norm": 0.467227142527021,
"kl": 0.9345703125,
"learning_rate": 1e-06,
"loss": 0.0079,
"reward": 0.7723214477300644,
"reward_std": 0.06612720899283886,
"rewards/unified_reward_func": 0.7723214477300644,
"step": 155
},
{
"clip_ratio": 0.0014949928154237568,
"epoch": 25.82758620689655,
"grad_norm": 4.261661247612953,
"kl": 0.7138671875,
"learning_rate": 1e-06,
"loss": 0.0073,
"step": 156
},
{
"batch_accuracy": 0.8660714285714286,
"clip_ratio": 0.0,
"completion_length": 240.62054443359375,
"epoch": 26.137931034482758,
"grad_norm": 14.49308916720133,
"kl": 4.377197265625,
"learning_rate": 1e-06,
"loss": 0.0017,
"reward": 0.8660714775323868,
"reward_std": 0.03562259301543236,
"rewards/unified_reward_func": 0.8660714775323868,
"step": 157
},
{
"clip_ratio": 0.0007656059751752764,
"epoch": 26.275862068965516,
"grad_norm": 0.34782130929462224,
"kl": 0.405517578125,
"learning_rate": 1e-06,
"loss": -0.0021,
"step": 158
},
{
"batch_accuracy": 0.8080357142857142,
"clip_ratio": 0.0,
"completion_length": 284.0803756713867,
"epoch": 26.413793103448278,
"grad_norm": 0.853993640481723,
"kl": 0.525390625,
"learning_rate": 1e-06,
"loss": 0.0448,
"reward": 0.808035746216774,
"reward_std": 0.12731034867465496,
"rewards/unified_reward_func": 0.808035746216774,
"step": 159
},
{
"clip_ratio": 0.0020131226046942174,
"epoch": 26.551724137931036,
"grad_norm": 0.6286474424499364,
"kl": 0.51904296875,
"learning_rate": 1e-06,
"loss": 0.0435,
"step": 160
},
{
"batch_accuracy": 0.7901785714285714,
"clip_ratio": 0.0,
"completion_length": 237.5357322692871,
"epoch": 26.689655172413794,
"grad_norm": 0.8611827252260262,
"kl": 0.66796875,
"learning_rate": 1e-06,
"loss": 0.0104,
"reward": 0.7901785969734192,
"reward_std": 0.07094572857022285,
"rewards/unified_reward_func": 0.7901785969734192,
"step": 161
},
{
"clip_ratio": 0.001519633864518255,
"epoch": 26.82758620689655,
"grad_norm": 0.44506417251011193,
"kl": 0.4794921875,
"learning_rate": 1e-06,
"loss": 0.0093,
"step": 162
},
{
"batch_accuracy": 0.7410714285714286,
"clip_ratio": 0.0,
"completion_length": 270.714298248291,
"epoch": 27.137931034482758,
"grad_norm": 0.6713248005427243,
"kl": 0.466796875,
"learning_rate": 1e-06,
"loss": 0.0146,
"reward": 0.7410714626312256,
"reward_std": 0.07875411584973335,
"rewards/unified_reward_func": 0.7410714626312256,
"step": 163
},
{
"clip_ratio": 0.001083969953469932,
"epoch": 27.275862068965516,
"grad_norm": 0.4707943331917379,
"kl": 0.5361328125,
"learning_rate": 1e-06,
"loss": 0.0135,
"step": 164
},
{
"batch_accuracy": 0.7901785714285714,
"clip_ratio": 0.0,
"completion_length": 239.80805206298828,
"epoch": 27.413793103448278,
"grad_norm": 12.949618265843812,
"kl": 12.4248046875,
"learning_rate": 1e-06,
"loss": 0.0007,
"reward": 0.790178582072258,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.790178582072258,
"step": 165
},
{
"clip_ratio": 0.0008150502981152385,
"epoch": 27.551724137931036,
"grad_norm": 1.2218985017147648,
"kl": 1.4560546875,
"learning_rate": 1e-06,
"loss": -0.0102,
"step": 166
},
{
"batch_accuracy": 0.90625,
"clip_ratio": 0.0,
"completion_length": 237.28572463989258,
"epoch": 27.689655172413794,
"grad_norm": 0.336672440220934,
"kl": 0.643798828125,
"learning_rate": 1e-06,
"loss": -0.0004,
"reward": 0.9062500447034836,
"reward_std": 0.03501640260219574,
"rewards/unified_reward_func": 0.9062500447034836,
"step": 167
},
{
"clip_ratio": 0.0004734238755190745,
"epoch": 27.82758620689655,
"grad_norm": 0.2530434744777561,
"kl": 0.53564453125,
"learning_rate": 1e-06,
"loss": -0.0009,
"step": 168
},
{
"batch_accuracy": 0.8705357142857143,
"clip_ratio": 0.0,
"completion_length": 239.7187614440918,
"epoch": 28.137931034482758,
"grad_norm": 0.43116339598093434,
"kl": 0.525390625,
"learning_rate": 1e-06,
"loss": 0.0053,
"reward": 0.870535746216774,
"reward_std": 0.05441322363913059,
"rewards/unified_reward_func": 0.870535746216774,
"step": 169
},
{
"clip_ratio": 0.0006829075573477894,
"epoch": 28.275862068965516,
"grad_norm": 0.2890658345945106,
"kl": 0.48779296875,
"learning_rate": 1e-06,
"loss": 0.0046,
"step": 170
},
{
"batch_accuracy": 0.7142857142857143,
"clip_ratio": 0.0,
"completion_length": 253.49108123779297,
"epoch": 28.413793103448278,
"grad_norm": 1.594006469856142,
"kl": 1.815673828125,
"learning_rate": 1e-06,
"loss": 0.0126,
"reward": 0.7142857611179352,
"reward_std": 0.05215509608387947,
"rewards/unified_reward_func": 0.7142857611179352,
"step": 171
},
{
"clip_ratio": 0.0009787115559447557,
"epoch": 28.551724137931036,
"grad_norm": 0.43736705326513975,
"kl": 0.75537109375,
"learning_rate": 1e-06,
"loss": 0.0114,
"step": 172
},
{
"batch_accuracy": 0.8616071428571429,
"clip_ratio": 0.0,
"completion_length": 306.3705520629883,
"epoch": 28.689655172413794,
"grad_norm": 5.999880286356358,
"kl": 3.625,
"learning_rate": 1e-06,
"loss": 0.0463,
"reward": 0.8616071790456772,
"reward_std": 0.11784721724689007,
"rewards/unified_reward_func": 0.8616071790456772,
"step": 173
},
{
"clip_ratio": 0.0030182256596162915,
"epoch": 28.82758620689655,
"grad_norm": 0.809363029282151,
"kl": 1.0537109375,
"learning_rate": 1e-06,
"loss": 0.0438,
"step": 174
},
{
"batch_accuracy": 0.7321428571428571,
"clip_ratio": 0.0,
"completion_length": 260.3750114440918,
"epoch": 29.137931034482758,
"grad_norm": 0.5032951734651111,
"kl": 0.41552734375,
"learning_rate": 1e-06,
"loss": -0.0038,
"reward": 0.73214291036129,
"reward_std": 0.0754547119140625,
"rewards/unified_reward_func": 0.73214291036129,
"step": 175
},
{
"clip_ratio": 0.0010940766078419983,
"epoch": 29.275862068965516,
"grad_norm": 0.3812137479874885,
"kl": 0.50537109375,
"learning_rate": 1e-06,
"loss": -0.0044,
"step": 176
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 297.16518783569336,
"epoch": 29.413793103448278,
"grad_norm": 0.9027143472678566,
"kl": 2.0537109375,
"learning_rate": 1e-06,
"loss": 0.0154,
"reward": 0.848214328289032,
"reward_std": 0.11272924393415451,
"rewards/unified_reward_func": 0.848214328289032,
"step": 177
},
{
"clip_ratio": 0.0022016192961018533,
"epoch": 29.551724137931036,
"grad_norm": 0.655278520928742,
"kl": 1.58935546875,
"learning_rate": 1e-06,
"loss": 0.0141,
"step": 178
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 245.64286422729492,
"epoch": 29.689655172413794,
"grad_norm": 0.6982969760632683,
"kl": 0.9599609375,
"learning_rate": 1e-06,
"loss": 0.0229,
"reward": 0.848214328289032,
"reward_std": 0.05050762742757797,
"rewards/unified_reward_func": 0.848214328289032,
"step": 179
},
{
"clip_ratio": 0.0013622870174003765,
"epoch": 29.82758620689655,
"grad_norm": 0.521585705506308,
"kl": 0.884765625,
"learning_rate": 1e-06,
"loss": 0.0221,
"step": 180
},
{
"batch_accuracy": 0.8080357142857142,
"clip_ratio": 0.0,
"completion_length": 283.63393783569336,
"epoch": 30.137931034482758,
"grad_norm": 0.6987516623234656,
"kl": 0.95458984375,
"learning_rate": 1e-06,
"loss": 0.0015,
"reward": 0.8080357611179352,
"reward_std": 0.070945730432868,
"rewards/unified_reward_func": 0.8080357611179352,
"step": 181
},
{
"clip_ratio": 0.001272890789550729,
"epoch": 30.275862068965516,
"grad_norm": 1.2679452071601345,
"kl": 0.7724609375,
"learning_rate": 1e-06,
"loss": 0.0012,
"step": 182
},
{
"batch_accuracy": 0.8705357142857143,
"clip_ratio": 0.0,
"completion_length": 216.00000762939453,
"epoch": 30.413793103448278,
"grad_norm": 0.23636473762762067,
"kl": 0.317626953125,
"learning_rate": 1e-06,
"loss": -0.0025,
"reward": 0.8705357611179352,
"reward_std": 0.018483899533748627,
"rewards/unified_reward_func": 0.8705357611179352,
"step": 183
},
{
"clip_ratio": 0.00020213250536471605,
"epoch": 30.551724137931036,
"grad_norm": 0.2376975480219316,
"kl": 0.305419921875,
"learning_rate": 1e-06,
"loss": -0.0028,
"step": 184
},
{
"batch_accuracy": 0.7946428571428572,
"clip_ratio": 0.0,
"completion_length": 237.61608123779297,
"epoch": 30.689655172413794,
"grad_norm": 0.9304541510624866,
"kl": 1.8798828125,
"learning_rate": 1e-06,
"loss": 0.0128,
"reward": 0.7946428954601288,
"reward_std": 0.056364621967077255,
"rewards/unified_reward_func": 0.7946428954601288,
"step": 185
},
{
"clip_ratio": 0.001274522306630388,
"epoch": 30.82758620689655,
"grad_norm": 0.6160668926137608,
"kl": 1.09228515625,
"learning_rate": 1e-06,
"loss": 0.012,
"step": 186
},
{
"batch_accuracy": 0.7857142857142857,
"clip_ratio": 0.0,
"completion_length": 264.026798248291,
"epoch": 31.137931034482758,
"grad_norm": 0.8381457440700479,
"kl": 0.68115234375,
"learning_rate": 1e-06,
"loss": 0.0165,
"reward": 0.7857143133878708,
"reward_std": 0.08357263542711735,
"rewards/unified_reward_func": 0.7857143133878708,
"step": 187
},
{
"clip_ratio": 0.0035605556913651526,
"epoch": 31.275862068965516,
"grad_norm": 1.1856915368204115,
"kl": 0.740234375,
"learning_rate": 1e-06,
"loss": 0.0153,
"step": 188
},
{
"batch_accuracy": 0.9241071428571428,
"clip_ratio": 0.0,
"completion_length": 252.27679443359375,
"epoch": 31.413793103448278,
"grad_norm": 0.6661533926688221,
"kl": 0.9140625,
"learning_rate": 1e-06,
"loss": -0.0026,
"reward": 0.9241071939468384,
"reward_std": 0.0689915269613266,
"rewards/unified_reward_func": 0.9241071939468384,
"step": 189
},
{
"clip_ratio": 0.0012773344933521003,
"epoch": 31.551724137931036,
"grad_norm": 0.9096020542402118,
"kl": 0.81103515625,
"learning_rate": 1e-06,
"loss": -0.0031,
"step": 190
},
{
"batch_accuracy": 0.75,
"clip_ratio": 0.0,
"completion_length": 231.93750762939453,
"epoch": 31.689655172413794,
"grad_norm": 0.5985491059230904,
"kl": 1.140625,
"learning_rate": 1e-06,
"loss": 0.0038,
"reward": 0.7500000447034836,
"reward_std": 0.08070831745862961,
"rewards/unified_reward_func": 0.7500000447034836,
"step": 191
},
{
"clip_ratio": 0.001351288432488218,
"epoch": 31.82758620689655,
"grad_norm": 0.43023596336510533,
"kl": 1.08935546875,
"learning_rate": 1e-06,
"loss": 0.0028,
"step": 192
},
{
"batch_accuracy": 0.8125,
"clip_ratio": 0.0,
"completion_length": 250.65179443359375,
"epoch": 32.13793103448276,
"grad_norm": 2.478437026215032,
"kl": 2.31494140625,
"learning_rate": 1e-06,
"loss": 0.0045,
"reward": 0.8125000298023224,
"reward_std": 0.05050762556493282,
"rewards/unified_reward_func": 0.8125000298023224,
"step": 193
},
{
"clip_ratio": 0.0017007194110192358,
"epoch": 32.275862068965516,
"grad_norm": 77.17716361138062,
"kl": 1.08447265625,
"learning_rate": 1e-06,
"loss": 0.0218,
"step": 194
},
{
"batch_accuracy": 0.8303571428571428,
"clip_ratio": 0.0,
"completion_length": 234.59376525878906,
"epoch": 32.41379310344828,
"grad_norm": 6.966001947932676,
"kl": 1.583984375,
"learning_rate": 1e-06,
"loss": 0.0106,
"reward": 0.8303571790456772,
"reward_std": 0.04959750920534134,
"rewards/unified_reward_func": 0.8303571790456772,
"step": 195
},
{
"clip_ratio": 0.0010828501835931093,
"epoch": 32.55172413793103,
"grad_norm": 0.41210893717986125,
"kl": 0.559326171875,
"learning_rate": 1e-06,
"loss": 0.0097,
"step": 196
},
{
"batch_accuracy": 0.8705357142857143,
"clip_ratio": 0.0,
"completion_length": 238.99108123779297,
"epoch": 32.689655172413794,
"grad_norm": 0.5727595579664349,
"kl": 0.84423828125,
"learning_rate": 1e-06,
"loss": 0.0,
"reward": 0.870535746216774,
"reward_std": 0.04569191299378872,
"rewards/unified_reward_func": 0.870535746216774,
"step": 197
},
{
"clip_ratio": 0.0004849687684327364,
"epoch": 32.827586206896555,
"grad_norm": 0.34508485808239964,
"kl": 0.692138671875,
"learning_rate": 1e-06,
"loss": -0.0008,
"step": 198
},
{
"batch_accuracy": 0.8660714285714286,
"clip_ratio": 0.0,
"completion_length": 233.93750762939453,
"epoch": 33.13793103448276,
"grad_norm": 0.6389097363094803,
"kl": 0.79345703125,
"learning_rate": 1e-06,
"loss": 0.0107,
"reward": 0.8660714775323868,
"reward_std": 0.056364623829722404,
"rewards/unified_reward_func": 0.8660714775323868,
"step": 199
},
{
"epoch": 33.275862068965516,
"grad_norm": 0.4172479026848871,
"learning_rate": 1e-06,
"loss": 0.0098,
"step": 200
},
{
"epoch": 33.275862068965516,
"eval_batch_accuracy": 0.7095238095238094,
"eval_clip_ratio": 0.0,
"eval_completion_length": 253.85001017252605,
"eval_kl": 0.13782552083333333,
"eval_loss": 0.012882467359304428,
"eval_reward": 0.7095238447189331,
"eval_reward_std": 0.14272718926270803,
"eval_rewards/unified_reward_func": 0.7095238447189331,
"eval_runtime": 309.75,
"eval_samples_per_second": 0.323,
"eval_steps_per_second": 0.006,
"step": 200
},
{
"batch_accuracy": 0.8125,
"clip_ratio": 0.00047189262841129676,
"completion_length": 216.61608123779297,
"epoch": 33.41379310344828,
"grad_norm": 0.34075034379789443,
"kl": 0.3455810546875,
"learning_rate": 1e-06,
"loss": 0.0028,
"reward": 0.8125000298023224,
"reward_std": 0.0417863167822361,
"rewards/unified_reward_func": 0.8125000298023224,
"step": 201
},
{
"clip_ratio": 0.0005730669654440135,
"epoch": 33.55172413793103,
"grad_norm": 0.24516084654318515,
"kl": 0.108154296875,
"learning_rate": 1e-06,
"loss": 0.0023,
"step": 202
},
{
"batch_accuracy": 0.8258928571428571,
"clip_ratio": 0.0,
"completion_length": 247.77679443359375,
"epoch": 33.689655172413794,
"grad_norm": 0.7986333842953461,
"kl": 0.114990234375,
"learning_rate": 1e-06,
"loss": 0.0313,
"reward": 0.8258928805589676,
"reward_std": 0.060270216315984726,
"rewards/unified_reward_func": 0.8258928805589676,
"step": 203
},
{
"clip_ratio": 0.0017704838537611067,
"epoch": 33.827586206896555,
"grad_norm": 0.5264166409776956,
"kl": 0.1175537109375,
"learning_rate": 1e-06,
"loss": 0.0304,
"step": 204
},
{
"batch_accuracy": 0.7366071428571428,
"clip_ratio": 0.0,
"completion_length": 239.7544822692871,
"epoch": 34.13793103448276,
"grad_norm": 0.5322234496834484,
"kl": 0.1153564453125,
"learning_rate": 1e-06,
"loss": -0.0007,
"reward": 0.736607164144516,
"reward_std": 0.07606089860200882,
"rewards/unified_reward_func": 0.736607164144516,
"step": 205
},
{
"clip_ratio": 0.0016890191909624264,
"epoch": 34.275862068965516,
"grad_norm": 0.37722043952212947,
"kl": 0.1268310546875,
"learning_rate": 1e-06,
"loss": -0.0017,
"step": 206
},
{
"batch_accuracy": 0.9017857142857143,
"clip_ratio": 0.0,
"completion_length": 260.91965103149414,
"epoch": 34.41379310344828,
"grad_norm": 0.6213850315224072,
"kl": 0.1051025390625,
"learning_rate": 1e-06,
"loss": 0.0085,
"reward": 0.9017857313156128,
"reward_std": 0.07576144114136696,
"rewards/unified_reward_func": 0.9017857313156128,
"step": 207
},
{
"clip_ratio": 0.00208102646865882,
"epoch": 34.55172413793103,
"grad_norm": 0.41446319845532803,
"kl": 0.11376953125,
"learning_rate": 1e-06,
"loss": 0.0079,
"step": 208
},
{
"batch_accuracy": 0.8080357142857143,
"clip_ratio": 0.0,
"completion_length": 220.27679443359375,
"epoch": 34.689655172413794,
"grad_norm": 0.34395769515755015,
"kl": 0.3416748046875,
"learning_rate": 1e-06,
"loss": 0.0033,
"reward": 0.8080357313156128,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.8080357313156128,
"step": 209
},
{
"clip_ratio": 0.0005923433491261676,
"epoch": 34.827586206896555,
"grad_norm": 0.24381267530999462,
"kl": 0.303466796875,
"learning_rate": 1e-06,
"loss": 0.0028,
"step": 210
},
{
"batch_accuracy": 0.8303571428571428,
"clip_ratio": 0.0,
"completion_length": 201.46429824829102,
"epoch": 35.13793103448276,
"grad_norm": 0.5610912818294096,
"kl": 0.1767578125,
"learning_rate": 1e-06,
"loss": 0.0062,
"reward": 0.8303571790456772,
"reward_std": 0.04764331132173538,
"rewards/unified_reward_func": 0.8303571790456772,
"step": 211
},
{
"clip_ratio": 0.002142434357665479,
"epoch": 35.275862068965516,
"grad_norm": 12.075891369202603,
"kl": 0.276123046875,
"learning_rate": 1e-06,
"loss": 0.0058,
"step": 212
},
{
"batch_accuracy": 0.71875,
"clip_ratio": 0.0,
"completion_length": 267.88393783569336,
"epoch": 35.41379310344828,
"grad_norm": 4.675264026045327,
"kl": 0.266845703125,
"learning_rate": 1e-06,
"loss": 0.0227,
"reward": 0.7187500447034836,
"reward_std": 0.060270220041275024,
"rewards/unified_reward_func": 0.7187500447034836,
"step": 213
},
{
"clip_ratio": 0.001382013550028205,
"epoch": 35.55172413793103,
"grad_norm": 3.166666135840778,
"kl": 0.205322265625,
"learning_rate": 1e-06,
"loss": 0.023,
"step": 214
},
{
"batch_accuracy": 0.8526785714285714,
"clip_ratio": 0.0,
"completion_length": 223.82143783569336,
"epoch": 35.689655172413794,
"grad_norm": 0.5784662868247091,
"kl": 0.2890625,
"learning_rate": 1e-06,
"loss": 0.0089,
"reward": 0.8526786118745804,
"reward_std": 0.08656250685453415,
"rewards/unified_reward_func": 0.8526786118745804,
"step": 215
},
{
"clip_ratio": 0.0020873736939392984,
"epoch": 35.827586206896555,
"grad_norm": 0.5514140064721655,
"kl": 0.25439453125,
"learning_rate": 1e-06,
"loss": 0.0083,
"step": 216
},
{
"batch_accuracy": 0.84375,
"clip_ratio": 0.0,
"completion_length": 216.41072463989258,
"epoch": 36.13793103448276,
"grad_norm": 0.5359392827868366,
"kl": 0.249267578125,
"learning_rate": 1e-06,
"loss": 0.0048,
"reward": 0.8437500298023224,
"reward_std": 0.05441322363913059,
"rewards/unified_reward_func": 0.8437500298023224,
"step": 217
},
{
"clip_ratio": 0.0010411773982923478,
"epoch": 36.275862068965516,
"grad_norm": 0.4041522386212472,
"kl": 0.2666015625,
"learning_rate": 1e-06,
"loss": 0.0042,
"step": 218
},
{
"batch_accuracy": 0.7678571428571429,
"clip_ratio": 0.0,
"completion_length": 235.70536422729492,
"epoch": 36.41379310344828,
"grad_norm": 59.360952385482854,
"kl": 1.32275390625,
"learning_rate": 1e-06,
"loss": -0.0075,
"reward": 0.7678571790456772,
"reward_std": 0.060876404866576195,
"rewards/unified_reward_func": 0.7678571790456772,
"step": 219
},
{
"clip_ratio": 0.003026856400538236,
"epoch": 36.55172413793103,
"grad_norm": 198.16594895250253,
"kl": 3.1953125,
"learning_rate": 1e-06,
"loss": -0.0052,
"step": 220
},
{
"batch_accuracy": 0.8080357142857143,
"clip_ratio": 0.0,
"completion_length": 215.12947463989258,
"epoch": 36.689655172413794,
"grad_norm": 1.110727771046951,
"kl": 0.7587890625,
"learning_rate": 1e-06,
"loss": 0.0131,
"reward": 0.8080357611179352,
"reward_std": 0.11377052217721939,
"rewards/unified_reward_func": 0.8080357611179352,
"step": 221
},
{
"clip_ratio": 0.0022536990436492488,
"epoch": 36.827586206896555,
"grad_norm": 1.226956128980916,
"kl": 0.72900390625,
"learning_rate": 1e-06,
"loss": 0.012,
"step": 222
},
{
"batch_accuracy": 0.8616071428571429,
"clip_ratio": 0.0,
"completion_length": 211.2812614440918,
"epoch": 37.13793103448276,
"grad_norm": 1.2105264518287366,
"kl": 0.985107421875,
"learning_rate": 1e-06,
"loss": 0.0147,
"reward": 0.861607164144516,
"reward_std": 0.056970810517668724,
"rewards/unified_reward_func": 0.861607164144516,
"step": 223
},
{
"clip_ratio": 0.001337285284535028,
"epoch": 37.275862068965516,
"grad_norm": 6.096977977454511,
"kl": 0.493896484375,
"learning_rate": 1e-06,
"loss": 0.0163,
"step": 224
},
{
"batch_accuracy": 0.9017857142857143,
"clip_ratio": 0.0,
"completion_length": 209.11608123779297,
"epoch": 37.41379310344828,
"grad_norm": 1.062706817262316,
"kl": 1.128662109375,
"learning_rate": 1e-06,
"loss": 0.0079,
"reward": 0.9017857611179352,
"reward_std": 0.04764330945909023,
"rewards/unified_reward_func": 0.9017857611179352,
"step": 225
},
{
"clip_ratio": 0.0014827463019173592,
"epoch": 37.55172413793103,
"grad_norm": 0.4654980038459283,
"kl": 0.870361328125,
"learning_rate": 1e-06,
"loss": 0.0071,
"step": 226
},
{
"batch_accuracy": 0.7991071428571428,
"clip_ratio": 0.0,
"completion_length": 204.71429443359375,
"epoch": 37.689655172413794,
"grad_norm": 2.0275758552802943,
"kl": 1.9033203125,
"learning_rate": 1e-06,
"loss": 0.011,
"reward": 0.799107164144516,
"reward_std": 0.05441322736442089,
"rewards/unified_reward_func": 0.799107164144516,
"step": 227
},
{
"clip_ratio": 0.0018361783877480775,
"epoch": 37.827586206896555,
"grad_norm": 0.6875113705883846,
"kl": 0.61181640625,
"learning_rate": 1e-06,
"loss": 0.0093,
"step": 228
},
{
"batch_accuracy": 0.9017857142857143,
"clip_ratio": 0.0,
"completion_length": 224.85268783569336,
"epoch": 38.13793103448276,
"grad_norm": 5.487262077964094,
"kl": 2.03857421875,
"learning_rate": 1e-06,
"loss": 0.0024,
"reward": 0.9017857313156128,
"reward_std": 0.04764331132173538,
"rewards/unified_reward_func": 0.9017857313156128,
"step": 229
},
{
"clip_ratio": 0.0015996035072021186,
"epoch": 38.275862068965516,
"grad_norm": 1450.0954313928476,
"kl": 0.55322265625,
"learning_rate": 1e-06,
"loss": 1.3522,
"step": 230
},
{
"batch_accuracy": 0.7857142857142857,
"clip_ratio": 0.0,
"completion_length": 225.71429443359375,
"epoch": 38.41379310344828,
"grad_norm": 0.6598845880184323,
"kl": 0.759765625,
"learning_rate": 1e-06,
"loss": -0.0111,
"reward": 0.785714328289032,
"reward_std": 0.0641758143901825,
"rewards/unified_reward_func": 0.785714328289032,
"step": 231
},
{
"clip_ratio": 0.001130485899921041,
"epoch": 38.55172413793103,
"grad_norm": 0.6240844125235968,
"kl": 0.95556640625,
"learning_rate": 1e-06,
"loss": -0.0118,
"step": 232
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 202.25447463989258,
"epoch": 38.689655172413794,
"grad_norm": 0.6091800162021634,
"kl": 0.791015625,
"learning_rate": 1e-06,
"loss": 0.0005,
"reward": 0.8839285969734192,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839285969734192,
"step": 233
},
{
"clip_ratio": 0.0003748063463717699,
"epoch": 38.827586206896555,
"grad_norm": 0.2670190645611779,
"kl": 0.51318359375,
"learning_rate": 1e-06,
"loss": -0.0002,
"step": 234
},
{
"batch_accuracy": 0.8705357142857143,
"clip_ratio": 0.0,
"completion_length": 197.33483123779297,
"epoch": 39.13793103448276,
"grad_norm": 8.036094604397608,
"kl": 6.68310546875,
"learning_rate": 1e-06,
"loss": 0.0196,
"reward": 0.870535746216774,
"reward_std": 0.04373771324753761,
"rewards/unified_reward_func": 0.870535746216774,
"step": 235
},
{
"clip_ratio": 0.0014344880473800004,
"epoch": 39.275862068965516,
"grad_norm": 1.3381116842000897,
"kl": 2.285400390625,
"learning_rate": 1e-06,
"loss": 0.0158,
"step": 236
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 246.4419708251953,
"epoch": 39.41379310344828,
"grad_norm": 0.7703391548316393,
"kl": 0.4609375,
"learning_rate": 1e-06,
"loss": 0.0111,
"reward": 0.8482143133878708,
"reward_std": 0.0774089116603136,
"rewards/unified_reward_func": 0.8482143133878708,
"step": 237
},
{
"clip_ratio": 0.0026967041776515543,
"epoch": 39.55172413793103,
"grad_norm": 0.6278183908739686,
"kl": 0.439453125,
"learning_rate": 1e-06,
"loss": 0.0107,
"step": 238
},
{
"batch_accuracy": 0.875,
"clip_ratio": 0.0,
"completion_length": 216.03125762939453,
"epoch": 39.689655172413794,
"grad_norm": 0.486258783542183,
"kl": 0.41552734375,
"learning_rate": 1e-06,
"loss": -0.0024,
"reward": 0.8750000298023224,
"reward_std": 0.033065006136894226,
"rewards/unified_reward_func": 0.8750000298023224,
"step": 239
},
{
"clip_ratio": 0.0005882278346689418,
"epoch": 39.827586206896555,
"grad_norm": 1.053399436053242,
"kl": 0.277587890625,
"learning_rate": 1e-06,
"loss": -0.0023,
"step": 240
},
{
"batch_accuracy": 0.8526785714285714,
"clip_ratio": 0.0,
"completion_length": 231.79019165039062,
"epoch": 40.13793103448276,
"grad_norm": 0.7411372005958455,
"kl": 0.4443359375,
"learning_rate": 1e-06,
"loss": 0.0114,
"reward": 0.8526785969734192,
"reward_std": 0.05441322363913059,
"rewards/unified_reward_func": 0.8526785969734192,
"step": 241
},
{
"clip_ratio": 0.0011974502704106271,
"epoch": 40.275862068965516,
"grad_norm": 0.5129553126144167,
"kl": 0.422119140625,
"learning_rate": 1e-06,
"loss": 0.0104,
"step": 242
},
{
"batch_accuracy": 0.8571428571428572,
"clip_ratio": 0.0,
"completion_length": 215.67858123779297,
"epoch": 40.41379310344828,
"grad_norm": 0.7342681774537219,
"kl": 0.75634765625,
"learning_rate": 1e-06,
"loss": -0.0114,
"reward": 0.8571428805589676,
"reward_std": 0.06417581252753735,
"rewards/unified_reward_func": 0.8571428805589676,
"step": 243
},
{
"clip_ratio": 0.00107419173582457,
"epoch": 40.55172413793103,
"grad_norm": 1.970726406090723,
"kl": 0.614013671875,
"learning_rate": 1e-06,
"loss": -0.0116,
"step": 244
},
{
"batch_accuracy": 0.8125,
"clip_ratio": 0.0,
"completion_length": 255.63840103149414,
"epoch": 40.689655172413794,
"grad_norm": 1.4462939125357268,
"kl": 0.83251953125,
"learning_rate": 1e-06,
"loss": 0.0111,
"reward": 0.8125000447034836,
"reward_std": 0.05050762742757797,
"rewards/unified_reward_func": 0.8125000447034836,
"step": 245
},
{
"clip_ratio": 0.0021176336449570954,
"epoch": 40.827586206896555,
"grad_norm": 0.6476801537145345,
"kl": 0.7197265625,
"learning_rate": 1e-06,
"loss": 0.0103,
"step": 246
},
{
"batch_accuracy": 0.9196428571428571,
"clip_ratio": 0.0,
"completion_length": 205.39733123779297,
"epoch": 41.13793103448276,
"grad_norm": 0.4995189751066373,
"kl": 0.55029296875,
"learning_rate": 1e-06,
"loss": 0.0013,
"reward": 0.9196428954601288,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9196428954601288,
"step": 247
},
{
"clip_ratio": 0.0004891223652521148,
"epoch": 41.275862068965516,
"grad_norm": 0.2501727453686479,
"kl": 0.30615234375,
"learning_rate": 1e-06,
"loss": 0.0008,
"step": 248
},
{
"batch_accuracy": 0.75,
"clip_ratio": 0.0,
"completion_length": 233.22768783569336,
"epoch": 41.41379310344828,
"grad_norm": 0.6614211017916019,
"kl": 0.5869140625,
"learning_rate": 1e-06,
"loss": 0.0071,
"reward": 0.7500000298023224,
"reward_std": 0.053500302135944366,
"rewards/unified_reward_func": 0.7500000298023224,
"step": 249
},
{
"epoch": 41.55172413793103,
"grad_norm": 0.5420823826241697,
"learning_rate": 1e-06,
"loss": 0.0067,
"step": 250
},
{
"epoch": 41.55172413793103,
"eval_batch_accuracy": 0.7309523809523809,
"eval_clip_ratio": 0.0,
"eval_completion_length": 242.77955525716146,
"eval_kl": 1.0609375,
"eval_loss": 0.01743003912270069,
"eval_reward": 0.7309524257977803,
"eval_reward_std": 0.14946034500996272,
"eval_rewards/unified_reward_func": 0.7309524257977803,
"eval_runtime": 300.631,
"eval_samples_per_second": 0.333,
"eval_steps_per_second": 0.007,
"step": 250
},
{
"batch_accuracy": 0.84375,
"clip_ratio": 0.0006140165933175012,
"completion_length": 239.43305206298828,
"epoch": 41.689655172413794,
"grad_norm": 10.36975411089091,
"kl": 4.4228515625,
"learning_rate": 1e-06,
"loss": 0.004,
"reward": 0.8437500596046448,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.8437500596046448,
"step": 251
},
{
"clip_ratio": 0.0005719505425076932,
"epoch": 41.827586206896555,
"grad_norm": 1.5376205862711856,
"kl": 0.7177734375,
"learning_rate": 1e-06,
"loss": -0.0016,
"step": 252
},
{
"batch_accuracy": 0.8794642857142857,
"clip_ratio": 0.0,
"completion_length": 239.58483505249023,
"epoch": 42.13793103448276,
"grad_norm": 0.4355553650409747,
"kl": 0.420166015625,
"learning_rate": 1e-06,
"loss": -0.004,
"reward": 0.8794643133878708,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.8794643133878708,
"step": 253
},
{
"clip_ratio": 0.0004529447469394654,
"epoch": 42.275862068965516,
"grad_norm": 0.3473121365753115,
"kl": 0.358154296875,
"learning_rate": 1e-06,
"loss": -0.0045,
"step": 254
},
{
"batch_accuracy": 0.7901785714285714,
"clip_ratio": 0.0,
"completion_length": 249.67858123779297,
"epoch": 42.41379310344828,
"grad_norm": 0.955097933412004,
"kl": 0.576171875,
"learning_rate": 1e-06,
"loss": 0.0245,
"reward": 0.7901785969734192,
"reward_std": 0.07966703735291958,
"rewards/unified_reward_func": 0.7901785969734192,
"step": 255
},
{
"clip_ratio": 0.0023401440121233463,
"epoch": 42.55172413793103,
"grad_norm": 0.6440396509406432,
"kl": 0.64599609375,
"learning_rate": 1e-06,
"loss": 0.0233,
"step": 256
},
{
"batch_accuracy": 0.8660714285714286,
"clip_ratio": 0.0,
"completion_length": 203.26340103149414,
"epoch": 42.689655172413794,
"grad_norm": 0.7450317244478006,
"kl": 0.638671875,
"learning_rate": 1e-06,
"loss": -0.0029,
"reward": 0.8660714626312256,
"reward_std": 0.06704013422131538,
"rewards/unified_reward_func": 0.8660714626312256,
"step": 257
},
{
"clip_ratio": 0.0013811402022838593,
"epoch": 42.827586206896555,
"grad_norm": 0.7976763809242704,
"kl": 0.36376953125,
"learning_rate": 1e-06,
"loss": -0.0034,
"step": 258
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 236.8303680419922,
"epoch": 43.13793103448276,
"grad_norm": 0.5496733412364695,
"kl": 0.44580078125,
"learning_rate": 1e-06,
"loss": 0.009,
"reward": 0.8482143431901932,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8482143431901932,
"step": 259
},
{
"clip_ratio": 0.0004358286823844537,
"epoch": 43.275862068965516,
"grad_norm": 0.4551862846969153,
"kl": 0.41845703125,
"learning_rate": 1e-06,
"loss": 0.0083,
"step": 260
},
{
"batch_accuracy": 0.8258928571428572,
"clip_ratio": 0.0,
"completion_length": 219.31251525878906,
"epoch": 43.41379310344828,
"grad_norm": 1.3480682509696915,
"kl": 1.4970703125,
"learning_rate": 1e-06,
"loss": 0.0263,
"reward": 0.8258928805589676,
"reward_std": 0.07966704294085503,
"rewards/unified_reward_func": 0.8258928805589676,
"step": 261
},
{
"clip_ratio": 0.0032285154156852514,
"epoch": 43.55172413793103,
"grad_norm": 1.3108389523737243,
"kl": 0.986328125,
"learning_rate": 1e-06,
"loss": 0.0257,
"step": 262
},
{
"batch_accuracy": 0.8660714285714286,
"clip_ratio": 0.0,
"completion_length": 250.83483505249023,
"epoch": 43.689655172413794,
"grad_norm": 1.5574408610060597,
"kl": 1.7861328125,
"learning_rate": 1e-06,
"loss": -0.0084,
"reward": 0.8660714626312256,
"reward_std": 0.11468344740569592,
"rewards/unified_reward_func": 0.8660714626312256,
"step": 263
},
{
"clip_ratio": 0.005030798492953181,
"epoch": 43.827586206896555,
"grad_norm": 4.383711031864911,
"kl": 1.2841796875,
"learning_rate": 1e-06,
"loss": -0.0092,
"step": 264
},
{
"batch_accuracy": 0.8392857142857143,
"clip_ratio": 0.0,
"completion_length": 214.05804824829102,
"epoch": 44.13793103448276,
"grad_norm": 1.3084821514107963,
"kl": 0.7490234375,
"learning_rate": 1e-06,
"loss": 0.0028,
"reward": 0.839285746216774,
"reward_std": 0.03111080639064312,
"rewards/unified_reward_func": 0.839285746216774,
"step": 265
},
{
"clip_ratio": 0.0014369665295816958,
"epoch": 44.275862068965516,
"grad_norm": 4.089177210944676,
"kl": 1.0859375,
"learning_rate": 1e-06,
"loss": 0.0029,
"step": 266
},
{
"batch_accuracy": 0.9196428571428571,
"clip_ratio": 0.0,
"completion_length": 230.01787185668945,
"epoch": 44.41379310344828,
"grad_norm": 8.798834640093343,
"kl": 5.603515625,
"learning_rate": 1e-06,
"loss": 0.0153,
"reward": 0.9196428954601288,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9196428954601288,
"step": 267
},
{
"clip_ratio": 0.001320267387200147,
"epoch": 44.55172413793103,
"grad_norm": 2.2251765859025703,
"kl": 2.681640625,
"learning_rate": 1e-06,
"loss": 0.0129,
"step": 268
},
{
"batch_accuracy": 0.8616071428571428,
"clip_ratio": 0.0,
"completion_length": 226.36608123779297,
"epoch": 44.689655172413794,
"grad_norm": 1940.7593108258543,
"kl": 1058.919921875,
"learning_rate": 1e-06,
"loss": 1.0647,
"reward": 0.861607164144516,
"reward_std": 0.07966703921556473,
"rewards/unified_reward_func": 0.861607164144516,
"step": 269
},
{
"clip_ratio": 0.0029754533316008747,
"epoch": 44.827586206896555,
"grad_norm": 838.2173848039702,
"kl": 5.6171875,
"learning_rate": 1e-06,
"loss": 0.2984,
"step": 270
},
{
"batch_accuracy": 0.8348214285714286,
"clip_ratio": 0.0,
"completion_length": 257.1026840209961,
"epoch": 45.13793103448276,
"grad_norm": 3.3428194911061606,
"kl": 2.40625,
"learning_rate": 1e-06,
"loss": -0.007,
"reward": 0.8348214477300644,
"reward_std": 0.03171699680387974,
"rewards/unified_reward_func": 0.8348214477300644,
"step": 271
},
{
"clip_ratio": 0.0009366457816213369,
"epoch": 45.275862068965516,
"grad_norm": 0.6784487520271747,
"kl": 1.31005859375,
"learning_rate": 1e-06,
"loss": -0.0081,
"step": 272
},
{
"batch_accuracy": 0.8392857142857143,
"clip_ratio": 0.0,
"completion_length": 222.82590866088867,
"epoch": 45.41379310344828,
"grad_norm": 0.5821113324215218,
"kl": 0.99609375,
"learning_rate": 1e-06,
"loss": 0.0089,
"reward": 0.839285746216774,
"reward_std": 0.03111080639064312,
"rewards/unified_reward_func": 0.839285746216774,
"step": 273
},
{
"clip_ratio": 0.0006448913627536967,
"epoch": 45.55172413793103,
"grad_norm": 0.5386933478140308,
"kl": 0.67138671875,
"learning_rate": 1e-06,
"loss": 0.0086,
"step": 274
},
{
"batch_accuracy": 0.8258928571428571,
"clip_ratio": 0.0,
"completion_length": 240.4196548461914,
"epoch": 45.689655172413794,
"grad_norm": 2.679436922054985,
"kl": 2.19140625,
"learning_rate": 1e-06,
"loss": 0.0119,
"reward": 0.8258928954601288,
"reward_std": 0.0709457267075777,
"rewards/unified_reward_func": 0.8258928954601288,
"step": 275
},
{
"clip_ratio": 0.0021198716713115573,
"epoch": 45.827586206896555,
"grad_norm": 0.7430334891757607,
"kl": 1.31494140625,
"learning_rate": 1e-06,
"loss": 0.0109,
"step": 276
},
{
"batch_accuracy": 0.8526785714285714,
"clip_ratio": 0.0,
"completion_length": 242.66965103149414,
"epoch": 46.13793103448276,
"grad_norm": 0.6689116919401598,
"kl": 0.66943359375,
"learning_rate": 1e-06,
"loss": 0.0053,
"reward": 0.8526785969734192,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.8526785969734192,
"step": 277
},
{
"clip_ratio": 0.0001296815025852993,
"epoch": 46.275862068965516,
"grad_norm": 0.48746993806188527,
"kl": 0.54248046875,
"learning_rate": 1e-06,
"loss": 0.0049,
"step": 278
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 247.94644165039062,
"epoch": 46.41379310344828,
"grad_norm": 0.8025427426991367,
"kl": 1.60498046875,
"learning_rate": 1e-06,
"loss": 0.0044,
"reward": 0.8482143431901932,
"reward_std": 0.07740890793502331,
"rewards/unified_reward_func": 0.8482143431901932,
"step": 279
},
{
"clip_ratio": 0.0018109494121745229,
"epoch": 46.55172413793103,
"grad_norm": 0.717943396090161,
"kl": 0.910400390625,
"learning_rate": 1e-06,
"loss": 0.0036,
"step": 280
},
{
"batch_accuracy": 0.8571428571428571,
"clip_ratio": 0.0,
"completion_length": 220.4687614440918,
"epoch": 46.689655172413794,
"grad_norm": 1.0984351537808166,
"kl": 1.27294921875,
"learning_rate": 1e-06,
"loss": -0.0036,
"reward": 0.85714291036129,
"reward_std": 0.050200898200273514,
"rewards/unified_reward_func": 0.85714291036129,
"step": 281
},
{
"clip_ratio": 0.0013987061684019864,
"epoch": 46.827586206896555,
"grad_norm": 0.8855299850661535,
"kl": 0.6982421875,
"learning_rate": 1e-06,
"loss": -0.004,
"step": 282
},
{
"batch_accuracy": 0.78125,
"clip_ratio": 0.0,
"completion_length": 237.95091247558594,
"epoch": 47.13793103448276,
"grad_norm": 0.7861590293685551,
"kl": 0.663330078125,
"learning_rate": 1e-06,
"loss": 0.0085,
"reward": 0.7812500298023224,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.7812500298023224,
"step": 283
},
{
"clip_ratio": 0.00020263424084987491,
"epoch": 47.275862068965516,
"grad_norm": 0.2348880465744704,
"kl": 0.33935546875,
"learning_rate": 1e-06,
"loss": 0.0078,
"step": 284
},
{
"batch_accuracy": 0.8348214285714286,
"clip_ratio": 0.0,
"completion_length": 236.69197845458984,
"epoch": 47.41379310344828,
"grad_norm": 0.6337068738976088,
"kl": 0.44580078125,
"learning_rate": 1e-06,
"loss": -0.0102,
"reward": 0.8348214626312256,
"reward_std": 0.05441322550177574,
"rewards/unified_reward_func": 0.8348214626312256,
"step": 285
},
{
"clip_ratio": 0.0008038270461838692,
"epoch": 47.55172413793103,
"grad_norm": 0.357288483372075,
"kl": 0.351806640625,
"learning_rate": 1e-06,
"loss": -0.0108,
"step": 286
},
{
"batch_accuracy": 0.9151785714285714,
"clip_ratio": 0.0,
"completion_length": 223.44644165039062,
"epoch": 47.689655172413794,
"grad_norm": 0.5834032629209837,
"kl": 0.589599609375,
"learning_rate": 1e-06,
"loss": -0.001,
"reward": 0.9151786118745804,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.9151786118745804,
"step": 287
},
{
"clip_ratio": 0.0004325698537286371,
"epoch": 47.827586206896555,
"grad_norm": 0.3715274484539023,
"kl": 0.354736328125,
"learning_rate": 1e-06,
"loss": -0.0017,
"step": 288
},
{
"batch_accuracy": 0.9107142857142857,
"clip_ratio": 0.0,
"completion_length": 247.15179061889648,
"epoch": 48.13793103448276,
"grad_norm": 0.8663466520277239,
"kl": 0.443359375,
"learning_rate": 1e-06,
"loss": 0.0129,
"reward": 0.9107143133878708,
"reward_std": 0.0417863167822361,
"rewards/unified_reward_func": 0.9107143133878708,
"step": 289
},
{
"clip_ratio": 0.0013505922979675233,
"epoch": 48.275862068965516,
"grad_norm": 0.47655967451195913,
"kl": 0.32861328125,
"learning_rate": 1e-06,
"loss": 0.0122,
"step": 290
},
{
"batch_accuracy": 0.8035714285714286,
"clip_ratio": 0.0,
"completion_length": 243.96429443359375,
"epoch": 48.41379310344828,
"grad_norm": 0.6368598067476341,
"kl": 1.74658203125,
"learning_rate": 1e-06,
"loss": 0.0096,
"reward": 0.8035714626312256,
"reward_std": 0.03111080639064312,
"rewards/unified_reward_func": 0.8035714626312256,
"step": 291
},
{
"clip_ratio": 0.0009445616160519421,
"epoch": 48.55172413793103,
"grad_norm": 0.3813073979718291,
"kl": 0.836181640625,
"learning_rate": 1e-06,
"loss": 0.0089,
"step": 292
},
{
"batch_accuracy": 0.9017857142857143,
"clip_ratio": 0.0,
"completion_length": 216.31251525878906,
"epoch": 48.689655172413794,
"grad_norm": 0.565064019599756,
"kl": 0.4580078125,
"learning_rate": 1e-06,
"loss": -0.008,
"reward": 0.901785746216774,
"reward_std": 0.056364621967077255,
"rewards/unified_reward_func": 0.901785746216774,
"step": 293
},
{
"clip_ratio": 0.0011924125137738883,
"epoch": 48.827586206896555,
"grad_norm": 0.3460829859213167,
"kl": 0.314208984375,
"learning_rate": 1e-06,
"loss": -0.0086,
"step": 294
},
{
"batch_accuracy": 0.9196428571428572,
"clip_ratio": 0.0,
"completion_length": 217.0937614440918,
"epoch": 49.13793103448276,
"grad_norm": 0.27367858227193687,
"kl": 0.396484375,
"learning_rate": 1e-06,
"loss": -0.0041,
"reward": 0.9196428656578064,
"reward_std": 0.016532503068447113,
"rewards/unified_reward_func": 0.9196428656578064,
"step": 295
},
{
"clip_ratio": 0.0003673049795906991,
"epoch": 49.275862068965516,
"grad_norm": 0.1876339620606552,
"kl": 0.421630859375,
"learning_rate": 1e-06,
"loss": -0.0044,
"step": 296
},
{
"batch_accuracy": 0.875,
"clip_ratio": 0.0,
"completion_length": 230.0446548461914,
"epoch": 49.41379310344828,
"grad_norm": 0.40936720132784676,
"kl": 0.369873046875,
"learning_rate": 1e-06,
"loss": -0.0005,
"reward": 0.8750000298023224,
"reward_std": 0.0417863167822361,
"rewards/unified_reward_func": 0.8750000298023224,
"step": 297
},
{
"clip_ratio": 0.0008250945247709751,
"epoch": 49.55172413793103,
"grad_norm": 0.30886124746802324,
"kl": 0.391357421875,
"learning_rate": 1e-06,
"loss": -0.001,
"step": 298
},
{
"batch_accuracy": 0.8348214285714286,
"clip_ratio": 0.0,
"completion_length": 215.32590866088867,
"epoch": 49.689655172413794,
"grad_norm": 1.2565852570926377,
"kl": 0.72021484375,
"learning_rate": 1e-06,
"loss": 0.0048,
"reward": 0.8348214626312256,
"reward_std": 0.04569191299378872,
"rewards/unified_reward_func": 0.8348214626312256,
"step": 299
},
{
"epoch": 49.827586206896555,
"grad_norm": 202.42169311010568,
"learning_rate": 1e-06,
"loss": 0.0688,
"step": 300
},
{
"epoch": 49.827586206896555,
"eval_batch_accuracy": 0.744047619047619,
"eval_clip_ratio": 0.0,
"eval_completion_length": 238.92111409505208,
"eval_kl": 0.09443359375,
"eval_loss": 0.007052370812743902,
"eval_reward": 0.7440476576487224,
"eval_reward_std": 0.1252529760201772,
"eval_rewards/unified_reward_func": 0.7440476576487224,
"eval_runtime": 295.3385,
"eval_samples_per_second": 0.339,
"eval_steps_per_second": 0.007,
"step": 300
},
{
"batch_accuracy": 0.8883928571428571,
"clip_ratio": 0.0004290657670935616,
"completion_length": 231.8526840209961,
"epoch": 50.13793103448276,
"grad_norm": 0.20282431054184938,
"kl": 0.21435546875,
"learning_rate": 1e-06,
"loss": -0.0047,
"reward": 0.8883928954601288,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.8883928954601288,
"step": 301
},
{
"clip_ratio": 5.189144212636165e-05,
"epoch": 50.275862068965516,
"grad_norm": 0.11943390673073986,
"kl": 0.081787109375,
"learning_rate": 1e-06,
"loss": -0.0049,
"step": 302
},
{
"batch_accuracy": 0.9196428571428572,
"clip_ratio": 0.0,
"completion_length": 231.25893783569336,
"epoch": 50.41379310344828,
"grad_norm": 0.3463004704729077,
"kl": 0.064453125,
"learning_rate": 1e-06,
"loss": -0.0109,
"reward": 0.9196428656578064,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9196428656578064,
"step": 303
},
{
"clip_ratio": 0.0006089565576985478,
"epoch": 50.55172413793103,
"grad_norm": 0.17425565901723739,
"kl": 0.06292724609375,
"learning_rate": 1e-06,
"loss": -0.0114,
"step": 304
},
{
"batch_accuracy": 0.7767857142857142,
"clip_ratio": 0.0,
"completion_length": 231.24108505249023,
"epoch": 50.689655172413794,
"grad_norm": 0.43862146523548484,
"kl": 0.1053466796875,
"learning_rate": 1e-06,
"loss": 0.005,
"reward": 0.7767857611179352,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.7767857611179352,
"step": 305
},
{
"clip_ratio": 0.0006555429135914892,
"epoch": 50.827586206896555,
"grad_norm": 0.24700514710664973,
"kl": 0.1103515625,
"learning_rate": 1e-06,
"loss": 0.0045,
"step": 306
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 233.2321548461914,
"epoch": 51.13793103448276,
"grad_norm": 0.39954111913085166,
"kl": 0.0909423828125,
"learning_rate": 1e-06,
"loss": -0.0034,
"reward": 0.8482143133878708,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8482143133878708,
"step": 307
},
{
"clip_ratio": 0.000625371903879568,
"epoch": 51.275862068965516,
"grad_norm": 0.17080867054188612,
"kl": 0.0911865234375,
"learning_rate": 1e-06,
"loss": -0.004,
"step": 308
},
{
"batch_accuracy": 0.7946428571428571,
"clip_ratio": 0.0,
"completion_length": 280.04019927978516,
"epoch": 51.41379310344828,
"grad_norm": 0.47116099169757397,
"kl": 0.197998046875,
"learning_rate": 1e-06,
"loss": -0.0044,
"reward": 0.7946428954601288,
"reward_std": 0.05831881985068321,
"rewards/unified_reward_func": 0.7946428954601288,
"step": 309
},
{
"clip_ratio": 0.0012788574094884098,
"epoch": 51.55172413793103,
"grad_norm": 0.30029225829463274,
"kl": 0.19482421875,
"learning_rate": 1e-06,
"loss": -0.005,
"step": 310
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 186.85715103149414,
"epoch": 51.689655172413794,
"grad_norm": 0.4418548735106828,
"kl": 0.100341796875,
"learning_rate": 1e-06,
"loss": 0.0075,
"reward": 0.8482143133878708,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8482143133878708,
"step": 311
},
{
"clip_ratio": 0.00036868505412712693,
"epoch": 51.827586206896555,
"grad_norm": 0.2913408678197493,
"kl": 0.1114501953125,
"learning_rate": 1e-06,
"loss": 0.0069,
"step": 312
},
{
"batch_accuracy": 0.8303571428571428,
"clip_ratio": 0.0,
"completion_length": 280.5044746398926,
"epoch": 52.13793103448276,
"grad_norm": 1.4439378816929274,
"kl": 0.15966796875,
"learning_rate": 1e-06,
"loss": 0.0152,
"reward": 0.8303571790456772,
"reward_std": 0.09229394607245922,
"rewards/unified_reward_func": 0.8303571790456772,
"step": 313
},
{
"clip_ratio": 0.00311722481274046,
"epoch": 52.275862068965516,
"grad_norm": 0.8123430006245371,
"kl": 0.193603515625,
"learning_rate": 1e-06,
"loss": 0.0141,
"step": 314
},
{
"batch_accuracy": 0.8571428571428571,
"clip_ratio": 0.0,
"completion_length": 222.40625381469727,
"epoch": 52.41379310344828,
"grad_norm": 0.04415438982672496,
"kl": 0.1348876953125,
"learning_rate": 1e-06,
"loss": 0.0001,
"reward": 0.8571428954601288,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.8571428954601288,
"step": 315
},
{
"clip_ratio": 0.0,
"epoch": 52.55172413793103,
"grad_norm": 0.04462437267095938,
"kl": 0.135498046875,
"learning_rate": 1e-06,
"loss": 0.0001,
"step": 316
},
{
"batch_accuracy": 0.9107142857142857,
"clip_ratio": 0.0,
"completion_length": 205.35715103149414,
"epoch": 52.689655172413794,
"grad_norm": 0.5442733579616569,
"kl": 0.173828125,
"learning_rate": 1e-06,
"loss": 0.0143,
"reward": 0.910714328289032,
"reward_std": 0.0417863167822361,
"rewards/unified_reward_func": 0.910714328289032,
"step": 317
},
{
"clip_ratio": 0.0013521471119020134,
"epoch": 52.827586206896555,
"grad_norm": 0.47979492418943703,
"kl": 0.226318359375,
"learning_rate": 1e-06,
"loss": 0.0136,
"step": 318
},
{
"batch_accuracy": 0.84375,
"clip_ratio": 0.0,
"completion_length": 231.1116180419922,
"epoch": 53.13793103448276,
"grad_norm": 0.9586342201910061,
"kl": 0.2431640625,
"learning_rate": 1e-06,
"loss": 0.0024,
"reward": 0.8437500447034836,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.8437500447034836,
"step": 319
},
{
"clip_ratio": 0.001583009579917416,
"epoch": 53.275862068965516,
"grad_norm": 0.4632957681239461,
"kl": 0.375244140625,
"learning_rate": 1e-06,
"loss": 0.0017,
"step": 320
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 242.8750114440918,
"epoch": 53.41379310344828,
"grad_norm": 1.084233760954395,
"kl": 0.63623046875,
"learning_rate": 1e-06,
"loss": -0.0007,
"reward": 0.8839286267757416,
"reward_std": 0.06704013235867023,
"rewards/unified_reward_func": 0.8839286267757416,
"step": 321
},
{
"clip_ratio": 0.002837361069396138,
"epoch": 53.55172413793103,
"grad_norm": 0.8171731868686768,
"kl": 0.8486328125,
"learning_rate": 1e-06,
"loss": -0.0015,
"step": 322
},
{
"batch_accuracy": 0.8169642857142857,
"clip_ratio": 0.0,
"completion_length": 245.22322463989258,
"epoch": 53.689655172413794,
"grad_norm": 0.4003946931627436,
"kl": 0.43408203125,
"learning_rate": 1e-06,
"loss": -0.0023,
"reward": 0.816964328289032,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.816964328289032,
"step": 323
},
{
"clip_ratio": 0.00021813277271576226,
"epoch": 53.827586206896555,
"grad_norm": 0.1998574216227568,
"kl": 0.41748046875,
"learning_rate": 1e-06,
"loss": -0.0026,
"step": 324
},
{
"batch_accuracy": 0.8660714285714286,
"clip_ratio": 0.0,
"completion_length": 305.17858123779297,
"epoch": 54.13793103448276,
"grad_norm": 0.6691039836172034,
"kl": 0.56787109375,
"learning_rate": 1e-06,
"loss": 0.0028,
"reward": 0.8660714626312256,
"reward_std": 0.056364621967077255,
"rewards/unified_reward_func": 0.8660714626312256,
"step": 325
},
{
"clip_ratio": 0.0009606677340343595,
"epoch": 54.275862068965516,
"grad_norm": 0.744355757067827,
"kl": 0.39453125,
"learning_rate": 1e-06,
"loss": 0.0023,
"step": 326
},
{
"batch_accuracy": 0.8080357142857143,
"clip_ratio": 0.0,
"completion_length": 227.95983123779297,
"epoch": 54.41379310344828,
"grad_norm": 0.6353132669066994,
"kl": 0.395751953125,
"learning_rate": 1e-06,
"loss": 0.0037,
"reward": 0.8080357313156128,
"reward_std": 0.04373771324753761,
"rewards/unified_reward_func": 0.8080357313156128,
"step": 327
},
{
"clip_ratio": 0.00077925888763275,
"epoch": 54.55172413793103,
"grad_norm": 0.5180919767758974,
"kl": 0.447509765625,
"learning_rate": 1e-06,
"loss": 0.003,
"step": 328
},
{
"batch_accuracy": 0.9196428571428572,
"clip_ratio": 0.0,
"completion_length": 226.0000114440918,
"epoch": 54.689655172413794,
"grad_norm": 0.40607137186611947,
"kl": 0.305419921875,
"learning_rate": 1e-06,
"loss": 0.0036,
"reward": 0.9196428656578064,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9196428656578064,
"step": 329
},
{
"clip_ratio": 0.0003793864598264918,
"epoch": 54.827586206896555,
"grad_norm": 0.29914570877843377,
"kl": 0.2880859375,
"learning_rate": 1e-06,
"loss": 0.003,
"step": 330
},
{
"batch_accuracy": 0.7321428571428571,
"clip_ratio": 0.0,
"completion_length": 280.06697845458984,
"epoch": 55.13793103448276,
"grad_norm": 0.8891067679376151,
"kl": 0.72314453125,
"learning_rate": 1e-06,
"loss": -0.0131,
"reward": 0.7321428805589676,
"reward_std": 0.05050762556493282,
"rewards/unified_reward_func": 0.7321428805589676,
"step": 331
},
{
"clip_ratio": 0.0006313109188340604,
"epoch": 55.275862068965516,
"grad_norm": 0.4728098972900883,
"kl": 0.37744140625,
"learning_rate": 1e-06,
"loss": -0.014,
"step": 332
},
{
"batch_accuracy": 0.8571428571428572,
"clip_ratio": 0.0,
"completion_length": 207.90625762939453,
"epoch": 55.41379310344828,
"grad_norm": 0.6588124033082572,
"kl": 0.29248046875,
"learning_rate": 1e-06,
"loss": -0.0149,
"reward": 0.8571428805589676,
"reward_std": 0.050200898200273514,
"rewards/unified_reward_func": 0.8571428805589676,
"step": 333
},
{
"clip_ratio": 0.0005623959586955607,
"epoch": 55.55172413793103,
"grad_norm": 0.9901216280739612,
"kl": 0.21533203125,
"learning_rate": 1e-06,
"loss": -0.0155,
"step": 334
},
{
"batch_accuracy": 0.8928571428571428,
"clip_ratio": 0.0,
"completion_length": 235.65179443359375,
"epoch": 55.689655172413794,
"grad_norm": 0.8589161497188839,
"kl": 0.422607421875,
"learning_rate": 1e-06,
"loss": 0.0113,
"reward": 0.8928571790456772,
"reward_std": 0.08161843568086624,
"rewards/unified_reward_func": 0.8928571790456772,
"step": 335
},
{
"clip_ratio": 0.0015647939726477489,
"epoch": 55.827586206896555,
"grad_norm": 286.42521108884824,
"kl": 0.23583984375,
"learning_rate": 1e-06,
"loss": 0.0848,
"step": 336
},
{
"batch_accuracy": 0.9910714285714286,
"clip_ratio": 0.0,
"completion_length": 221.1785774230957,
"epoch": 56.13793103448276,
"grad_norm": 0.47598821491528226,
"kl": 0.222900390625,
"learning_rate": 1e-06,
"loss": -0.0022,
"reward": 0.9910714626312256,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9910714626312256,
"step": 337
},
{
"clip_ratio": 0.0003239207435399294,
"epoch": 56.275862068965516,
"grad_norm": 0.26568970245571083,
"kl": 0.291259765625,
"learning_rate": 1e-06,
"loss": -0.0027,
"step": 338
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 230.15625762939453,
"epoch": 56.41379310344828,
"grad_norm": 4.759978452673977,
"kl": 1.978271484375,
"learning_rate": 1e-06,
"loss": -0.0017,
"reward": 0.8839286267757416,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839286267757416,
"step": 339
},
{
"clip_ratio": 0.00047356344293802977,
"epoch": 56.55172413793103,
"grad_norm": 0.48573771123545095,
"kl": 0.251220703125,
"learning_rate": 1e-06,
"loss": -0.0034,
"step": 340
},
{
"batch_accuracy": 0.6785714285714286,
"clip_ratio": 0.0,
"completion_length": 284.6651916503906,
"epoch": 56.689655172413794,
"grad_norm": 0.39208663367499275,
"kl": 0.23291015625,
"learning_rate": 1e-06,
"loss": -0.0069,
"reward": 0.6785714626312256,
"reward_std": 0.03696779906749725,
"rewards/unified_reward_func": 0.6785714626312256,
"step": 341
},
{
"clip_ratio": 0.00026187896582996473,
"epoch": 56.827586206896555,
"grad_norm": 0.312513512345737,
"kl": 0.2509765625,
"learning_rate": 1e-06,
"loss": -0.0076,
"step": 342
},
{
"batch_accuracy": 0.8883928571428571,
"clip_ratio": 0.0,
"completion_length": 243.75447845458984,
"epoch": 57.13793103448276,
"grad_norm": 24.52458761423547,
"kl": 6.576416015625,
"learning_rate": 1e-06,
"loss": 0.0041,
"reward": 0.8883928805589676,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.8883928805589676,
"step": 343
},
{
"clip_ratio": 0.0012561257171910256,
"epoch": 57.275862068965516,
"grad_norm": 0.534774035654984,
"kl": 0.287109375,
"learning_rate": 1e-06,
"loss": -0.0022,
"step": 344
},
{
"batch_accuracy": 0.8169642857142857,
"clip_ratio": 0.0,
"completion_length": 262.10268783569336,
"epoch": 57.41379310344828,
"grad_norm": 4.2828878662326595,
"kl": 2.083984375,
"learning_rate": 1e-06,
"loss": 0.0042,
"reward": 0.8169643431901932,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.8169643431901932,
"step": 345
},
{
"clip_ratio": 0.00011598970741033554,
"epoch": 57.55172413793103,
"grad_norm": 0.25471420002458783,
"kl": 0.4111328125,
"learning_rate": 1e-06,
"loss": 0.0026,
"step": 346
},
{
"batch_accuracy": 0.9241071428571428,
"clip_ratio": 0.0,
"completion_length": 216.49554061889648,
"epoch": 57.689655172413794,
"grad_norm": 0.40655736639617174,
"kl": 0.369873046875,
"learning_rate": 1e-06,
"loss": -0.0014,
"reward": 0.9241071939468384,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.9241071939468384,
"step": 347
},
{
"clip_ratio": 9.582215716363862e-05,
"epoch": 57.827586206896555,
"grad_norm": 0.20228576025102957,
"kl": 0.185302734375,
"learning_rate": 1e-06,
"loss": -0.0019,
"step": 348
},
{
"batch_accuracy": 0.9508928571428571,
"clip_ratio": 0.0,
"completion_length": 223.70983123779297,
"epoch": 58.13793103448276,
"grad_norm": 0.3934745552241851,
"kl": 0.159912109375,
"learning_rate": 1e-06,
"loss": 0.0105,
"reward": 0.9508928954601288,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.9508928954601288,
"step": 349
},
{
"epoch": 58.275862068965516,
"grad_norm": 0.47304722160177515,
"learning_rate": 1e-06,
"loss": 0.01,
"step": 350
},
{
"epoch": 58.275862068965516,
"eval_batch_accuracy": 0.7678571428571428,
"eval_clip_ratio": 0.0,
"eval_completion_length": 250.05552876790364,
"eval_kl": 0.5563151041666666,
"eval_loss": -0.0007301162695512176,
"eval_reward": 0.7678571780522664,
"eval_reward_std": 0.11819532414277395,
"eval_rewards/unified_reward_func": 0.7678571780522664,
"eval_runtime": 299.8452,
"eval_samples_per_second": 0.334,
"eval_steps_per_second": 0.007,
"step": 350
},
{
"batch_accuracy": 0.7946428571428571,
"clip_ratio": 0.0006118576420703903,
"completion_length": 244.39733123779297,
"epoch": 58.41379310344828,
"grad_norm": 0.3158750121263176,
"kl": 0.20098876953125,
"learning_rate": 1e-06,
"loss": 0.0033,
"reward": 0.7946428954601288,
"reward_std": 0.016532503068447113,
"rewards/unified_reward_func": 0.7946428954601288,
"step": 351
},
{
"clip_ratio": 0.00023174374655354768,
"epoch": 58.55172413793103,
"grad_norm": 0.2202251953260023,
"kl": 0.1893310546875,
"learning_rate": 1e-06,
"loss": 0.0029,
"step": 352
},
{
"batch_accuracy": 0.7946428571428572,
"clip_ratio": 0.0,
"completion_length": 243.59822463989258,
"epoch": 58.689655172413794,
"grad_norm": 0.6233607920605455,
"kl": 0.64697265625,
"learning_rate": 1e-06,
"loss": -0.0133,
"reward": 0.7946428954601288,
"reward_std": 0.05831881985068321,
"rewards/unified_reward_func": 0.7946428954601288,
"step": 353
},
{
"clip_ratio": 0.0008263020572485402,
"epoch": 58.827586206896555,
"grad_norm": 0.4362430398528479,
"kl": 0.450439453125,
"learning_rate": 1e-06,
"loss": -0.0139,
"step": 354
},
{
"batch_accuracy": 0.8794642857142857,
"clip_ratio": 0.0,
"completion_length": 228.5357208251953,
"epoch": 59.13793103448276,
"grad_norm": 0.37929784934812716,
"kl": 0.2900390625,
"learning_rate": 1e-06,
"loss": -0.0112,
"reward": 0.879464328289032,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.879464328289032,
"step": 355
},
{
"clip_ratio": 0.0007167806033976376,
"epoch": 59.275862068965516,
"grad_norm": 3.93509932542601,
"kl": 0.1904296875,
"learning_rate": 1e-06,
"loss": -0.0104,
"step": 356
},
{
"batch_accuracy": 0.9464285714285714,
"clip_ratio": 0.0,
"completion_length": 204.1250114440918,
"epoch": 59.41379310344828,
"grad_norm": 0.38429841165810424,
"kl": 0.166748046875,
"learning_rate": 1e-06,
"loss": 0.0011,
"reward": 0.9464285969734192,
"reward_std": 0.033065006136894226,
"rewards/unified_reward_func": 0.9464285969734192,
"step": 357
},
{
"clip_ratio": 0.0006046723137842491,
"epoch": 59.55172413793103,
"grad_norm": 0.33219320585174655,
"kl": 0.1893310546875,
"learning_rate": 1e-06,
"loss": 0.0007,
"step": 358
},
{
"batch_accuracy": 0.7633928571428571,
"clip_ratio": 0.0,
"completion_length": 262.7410888671875,
"epoch": 59.689655172413794,
"grad_norm": 0.6733029338519297,
"kl": 0.2568359375,
"learning_rate": 1e-06,
"loss": -0.015,
"reward": 0.7633928954601288,
"reward_std": 0.05441322363913059,
"rewards/unified_reward_func": 0.7633928954601288,
"step": 359
},
{
"clip_ratio": 0.0009442013979423791,
"epoch": 59.827586206896555,
"grad_norm": 0.5664337727116652,
"kl": 0.290283203125,
"learning_rate": 1e-06,
"loss": -0.0154,
"step": 360
},
{
"batch_accuracy": 0.9107142857142857,
"clip_ratio": 0.0,
"completion_length": 209.6651840209961,
"epoch": 60.13793103448276,
"grad_norm": 0.6726035076697437,
"kl": 0.287841796875,
"learning_rate": 1e-06,
"loss": -0.0003,
"reward": 0.910714328289032,
"reward_std": 0.0417863167822361,
"rewards/unified_reward_func": 0.910714328289032,
"step": 361
},
{
"clip_ratio": 0.0008161436417140067,
"epoch": 60.275862068965516,
"grad_norm": 0.4014200013710924,
"kl": 0.245849609375,
"learning_rate": 1e-06,
"loss": -0.0013,
"step": 362
},
{
"batch_accuracy": 0.9196428571428571,
"clip_ratio": 0.0,
"completion_length": 257.308048248291,
"epoch": 60.41379310344828,
"grad_norm": 0.5910338904683496,
"kl": 0.564208984375,
"learning_rate": 1e-06,
"loss": -0.0033,
"reward": 0.9196428954601288,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9196428954601288,
"step": 363
},
{
"clip_ratio": 0.0003132874990114942,
"epoch": 60.55172413793103,
"grad_norm": 0.1909713517011228,
"kl": 0.2470703125,
"learning_rate": 1e-06,
"loss": -0.0037,
"step": 364
},
{
"batch_accuracy": 0.7723214285714286,
"clip_ratio": 0.0,
"completion_length": 232.00447463989258,
"epoch": 60.689655172413794,
"grad_norm": 0.43041414128294087,
"kl": 0.199462890625,
"learning_rate": 1e-06,
"loss": 0.0035,
"reward": 0.7723214477300644,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.7723214477300644,
"step": 365
},
{
"clip_ratio": 0.0005480971594806761,
"epoch": 60.827586206896555,
"grad_norm": 0.2982648433527683,
"kl": 0.2138671875,
"learning_rate": 1e-06,
"loss": 0.0029,
"step": 366
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 193.9598274230957,
"epoch": 61.13793103448276,
"grad_norm": 143.25192413352647,
"kl": 55.736328125,
"learning_rate": 1e-06,
"loss": 0.0511,
"reward": 0.8839285969734192,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839285969734192,
"step": 367
},
{
"clip_ratio": 0.0007856006559450179,
"epoch": 61.275862068965516,
"grad_norm": 215.10922197761653,
"kl": 0.515625,
"learning_rate": 1e-06,
"loss": 0.1139,
"step": 368
},
{
"batch_accuracy": 0.8660714285714286,
"clip_ratio": 0.0,
"completion_length": 251.83929443359375,
"epoch": 61.41379310344828,
"grad_norm": 0.749264934206035,
"kl": 0.57666015625,
"learning_rate": 1e-06,
"loss": 0.0052,
"reward": 0.8660714775323868,
"reward_std": 0.05831881985068321,
"rewards/unified_reward_func": 0.8660714775323868,
"step": 369
},
{
"clip_ratio": 0.0008260020840680227,
"epoch": 61.55172413793103,
"grad_norm": 0.4966394103134508,
"kl": 0.284912109375,
"learning_rate": 1e-06,
"loss": 0.0041,
"step": 370
},
{
"batch_accuracy": 0.9241071428571428,
"clip_ratio": 0.0,
"completion_length": 231.1250114440918,
"epoch": 61.689655172413794,
"grad_norm": 0.43740040556746523,
"kl": 0.4306640625,
"learning_rate": 1e-06,
"loss": -0.0017,
"reward": 0.924107164144516,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.924107164144516,
"step": 371
},
{
"clip_ratio": 7.185972935985774e-05,
"epoch": 61.827586206896555,
"grad_norm": 0.23887790320111377,
"kl": 0.353515625,
"learning_rate": 1e-06,
"loss": -0.002,
"step": 372
},
{
"batch_accuracy": 0.7723214285714286,
"clip_ratio": 0.0,
"completion_length": 247.37054443359375,
"epoch": 62.13793103448276,
"grad_norm": 0.6206085533572123,
"kl": 0.36083984375,
"learning_rate": 1e-06,
"loss": 0.0151,
"reward": 0.7723214775323868,
"reward_std": 0.03788071870803833,
"rewards/unified_reward_func": 0.7723214775323868,
"step": 373
},
{
"clip_ratio": 0.001170877949334681,
"epoch": 62.275862068965516,
"grad_norm": 0.34783589826385763,
"kl": 0.228515625,
"learning_rate": 1e-06,
"loss": 0.0145,
"step": 374
},
{
"batch_accuracy": 0.8928571428571428,
"clip_ratio": 0.0,
"completion_length": 226.1250114440918,
"epoch": 62.41379310344828,
"grad_norm": 0.33943189999115525,
"kl": 0.429931640625,
"learning_rate": 1e-06,
"loss": 0.0004,
"reward": 0.8928571939468384,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.8928571939468384,
"step": 375
},
{
"clip_ratio": 0.0,
"epoch": 62.55172413793103,
"grad_norm": 0.04327261789576398,
"kl": 0.22802734375,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 376
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 240.47322845458984,
"epoch": 62.689655172413794,
"grad_norm": 0.506391391627102,
"kl": 0.338623046875,
"learning_rate": 1e-06,
"loss": 0.0074,
"reward": 0.8839285969734192,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839285969734192,
"step": 377
},
{
"clip_ratio": 0.0006104611675255001,
"epoch": 62.827586206896555,
"grad_norm": 0.2965824495255123,
"kl": 0.333251953125,
"learning_rate": 1e-06,
"loss": 0.0065,
"step": 378
},
{
"batch_accuracy": 0.9642857142857143,
"clip_ratio": 0.0,
"completion_length": 226.99108123779297,
"epoch": 63.13793103448276,
"grad_norm": 0.06203814627270833,
"kl": 0.3408203125,
"learning_rate": 1e-06,
"loss": 0.0003,
"reward": 0.9642857313156128,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.9642857313156128,
"step": 379
},
{
"clip_ratio": 0.0,
"epoch": 63.275862068965516,
"grad_norm": 0.045239787224193535,
"kl": 0.284912109375,
"learning_rate": 1e-06,
"loss": 0.0003,
"step": 380
},
{
"batch_accuracy": 0.7991071428571428,
"clip_ratio": 0.0,
"completion_length": 235.86608123779297,
"epoch": 63.41379310344828,
"grad_norm": 0.3996258785535166,
"kl": 0.344970703125,
"learning_rate": 1e-06,
"loss": -0.0054,
"reward": 0.7991071790456772,
"reward_std": 0.03501640260219574,
"rewards/unified_reward_func": 0.7991071790456772,
"step": 381
},
{
"clip_ratio": 0.0003395631938474253,
"epoch": 63.55172413793103,
"grad_norm": 0.3568904247413468,
"kl": 0.316162109375,
"learning_rate": 1e-06,
"loss": -0.006,
"step": 382
},
{
"batch_accuracy": 0.8616071428571428,
"clip_ratio": 0.0,
"completion_length": 212.32590103149414,
"epoch": 63.689655172413794,
"grad_norm": 0.6000615523479406,
"kl": 0.24365234375,
"learning_rate": 1e-06,
"loss": -0.0073,
"reward": 0.8616071790456772,
"reward_std": 0.060270218178629875,
"rewards/unified_reward_func": 0.8616071790456772,
"step": 383
},
{
"clip_ratio": 0.0009390746981807752,
"epoch": 63.827586206896555,
"grad_norm": 0.3933344938486858,
"kl": 0.2490234375,
"learning_rate": 1e-06,
"loss": -0.0082,
"step": 384
},
{
"batch_accuracy": 0.90625,
"clip_ratio": 0.0,
"completion_length": 250.05358505249023,
"epoch": 64.13793103448276,
"grad_norm": 0.8330463899634639,
"kl": 0.247802734375,
"learning_rate": 1e-06,
"loss": 0.01,
"reward": 0.9062500298023224,
"reward_std": 0.05441322550177574,
"rewards/unified_reward_func": 0.9062500298023224,
"step": 385
},
{
"clip_ratio": 0.0013701908756047487,
"epoch": 64.27586206896552,
"grad_norm": 0.47745481228963144,
"kl": 0.24755859375,
"learning_rate": 1e-06,
"loss": 0.009,
"step": 386
},
{
"batch_accuracy": 0.84375,
"clip_ratio": 0.0,
"completion_length": 218.30804824829102,
"epoch": 64.41379310344827,
"grad_norm": 0.4860256067494172,
"kl": 0.374267578125,
"learning_rate": 1e-06,
"loss": -0.0047,
"reward": 0.8437500298023224,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.8437500298023224,
"step": 387
},
{
"clip_ratio": 0.0006055501580704004,
"epoch": 64.55172413793103,
"grad_norm": 0.2341605321475626,
"kl": 0.273193359375,
"learning_rate": 1e-06,
"loss": -0.0051,
"step": 388
},
{
"batch_accuracy": 0.875,
"clip_ratio": 0.0,
"completion_length": 222.17858123779297,
"epoch": 64.6896551724138,
"grad_norm": 0.5574211120856241,
"kl": 0.27490234375,
"learning_rate": 1e-06,
"loss": -0.0048,
"reward": 0.8750000298023224,
"reward_std": 0.05050762742757797,
"rewards/unified_reward_func": 0.8750000298023224,
"step": 389
},
{
"clip_ratio": 0.0009339429816463962,
"epoch": 64.82758620689656,
"grad_norm": 0.514534910574207,
"kl": 0.236083984375,
"learning_rate": 1e-06,
"loss": -0.0053,
"step": 390
},
{
"batch_accuracy": 0.8392857142857143,
"clip_ratio": 0.0,
"completion_length": 228.56697845458984,
"epoch": 65.13793103448276,
"grad_norm": 0.5745364965962138,
"kl": 0.2763671875,
"learning_rate": 1e-06,
"loss": -0.0238,
"reward": 0.839285746216774,
"reward_std": 0.05050762742757797,
"rewards/unified_reward_func": 0.839285746216774,
"step": 391
},
{
"clip_ratio": 0.0009972895204555243,
"epoch": 65.27586206896552,
"grad_norm": 0.33961545508933666,
"kl": 0.28369140625,
"learning_rate": 1e-06,
"loss": -0.0245,
"step": 392
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 223.16518783569336,
"epoch": 65.41379310344827,
"grad_norm": 0.3434587742819441,
"kl": 0.2099609375,
"learning_rate": 1e-06,
"loss": -0.0005,
"reward": 0.8482143431901932,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8482143431901932,
"step": 393
},
{
"clip_ratio": 0.0007103000534698367,
"epoch": 65.55172413793103,
"grad_norm": 0.20376672707462493,
"kl": 0.248046875,
"learning_rate": 1e-06,
"loss": -0.0007,
"step": 394
},
{
"batch_accuracy": 0.9196428571428571,
"clip_ratio": 0.0,
"completion_length": 214.17858123779297,
"epoch": 65.6896551724138,
"grad_norm": 0.42867069975063926,
"kl": 0.181396484375,
"learning_rate": 1e-06,
"loss": 0.0072,
"reward": 0.9196428954601288,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9196428954601288,
"step": 395
},
{
"clip_ratio": 0.00042697125172708184,
"epoch": 65.82758620689656,
"grad_norm": 0.20848546236555793,
"kl": 0.192626953125,
"learning_rate": 1e-06,
"loss": 0.0067,
"step": 396
},
{
"batch_accuracy": 0.84375,
"clip_ratio": 0.0,
"completion_length": 208.79019165039062,
"epoch": 66.13793103448276,
"grad_norm": 0.5565175606934627,
"kl": 0.481201171875,
"learning_rate": 1e-06,
"loss": 0.0012,
"reward": 0.8437500298023224,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.8437500298023224,
"step": 397
},
{
"clip_ratio": 0.0008260089380200952,
"epoch": 66.27586206896552,
"grad_norm": 0.3524931608816777,
"kl": 0.369384765625,
"learning_rate": 1e-06,
"loss": 0.0004,
"step": 398
},
{
"batch_accuracy": 0.8660714285714286,
"clip_ratio": 0.0,
"completion_length": 250.0178680419922,
"epoch": 66.41379310344827,
"grad_norm": 0.4351526947834308,
"kl": 0.41015625,
"learning_rate": 1e-06,
"loss": -0.0251,
"reward": 0.8660714477300644,
"reward_std": 0.04434390366077423,
"rewards/unified_reward_func": 0.8660714477300644,
"step": 399
},
{
"epoch": 66.55172413793103,
"grad_norm": 0.24679700643125732,
"learning_rate": 1e-06,
"loss": -0.0257,
"step": 400
},
{
"epoch": 66.55172413793103,
"eval_batch_accuracy": 0.7511904761904762,
"eval_clip_ratio": 0.0,
"eval_completion_length": 248.59719746907552,
"eval_kl": 0.07989908854166666,
"eval_loss": 0.016887282952666283,
"eval_reward": 0.7511905193328857,
"eval_reward_std": 0.12925923814376195,
"eval_rewards/unified_reward_func": 0.7511905193328857,
"eval_runtime": 296.0887,
"eval_samples_per_second": 0.338,
"eval_steps_per_second": 0.007,
"step": 400
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.00014136406025500037,
"completion_length": 253.33037185668945,
"epoch": 66.6896551724138,
"grad_norm": 0.31349476916552094,
"kl": 0.20245361328125,
"learning_rate": 1e-06,
"loss": -0.0008,
"reward": 0.8839286267757416,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839286267757416,
"step": 401
},
{
"clip_ratio": 0.00036735970206791535,
"epoch": 66.82758620689656,
"grad_norm": 0.19451669827626714,
"kl": 0.0814208984375,
"learning_rate": 1e-06,
"loss": -0.0014,
"step": 402
},
{
"batch_accuracy": 0.8883928571428571,
"clip_ratio": 0.0,
"completion_length": 215.8839340209961,
"epoch": 67.13793103448276,
"grad_norm": 0.38417199755616704,
"kl": 0.0771484375,
"learning_rate": 1e-06,
"loss": 0.0008,
"reward": 0.8883928954601288,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.8883928954601288,
"step": 403
},
{
"clip_ratio": 8.53727906360291e-05,
"epoch": 67.27586206896552,
"grad_norm": 0.18756072191965203,
"kl": 0.081298828125,
"learning_rate": 1e-06,
"loss": 0.0003,
"step": 404
},
{
"batch_accuracy": 0.9598214285714286,
"clip_ratio": 0.0,
"completion_length": 225.5178680419922,
"epoch": 67.41379310344827,
"grad_norm": 0.3038476247906523,
"kl": 0.1053466796875,
"learning_rate": 1e-06,
"loss": 0.0025,
"reward": 0.9598214626312256,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.9598214626312256,
"step": 405
},
{
"clip_ratio": 0.00034553295699879527,
"epoch": 67.55172413793103,
"grad_norm": 0.17955543384066075,
"kl": 0.1016845703125,
"learning_rate": 1e-06,
"loss": 0.0021,
"step": 406
},
{
"batch_accuracy": 0.8303571428571429,
"clip_ratio": 0.0,
"completion_length": 266.62500381469727,
"epoch": 67.6896551724138,
"grad_norm": 0.40599740725102873,
"kl": 0.1082763671875,
"learning_rate": 1e-06,
"loss": -0.0176,
"reward": 0.830357164144516,
"reward_std": 0.056364621967077255,
"rewards/unified_reward_func": 0.830357164144516,
"step": 407
},
{
"clip_ratio": 0.0004739694340969436,
"epoch": 67.82758620689656,
"grad_norm": 0.32870338345791006,
"kl": 0.10009765625,
"learning_rate": 1e-06,
"loss": -0.0182,
"step": 408
},
{
"batch_accuracy": 0.9151785714285714,
"clip_ratio": 0.0,
"completion_length": 243.8616180419922,
"epoch": 68.13793103448276,
"grad_norm": 0.516708055153792,
"kl": 0.0828857421875,
"learning_rate": 1e-06,
"loss": 0.0014,
"reward": 0.9151785969734192,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.9151785969734192,
"step": 409
},
{
"clip_ratio": 0.00027526616759132594,
"epoch": 68.27586206896552,
"grad_norm": 0.28028763512050475,
"kl": 0.0921630859375,
"learning_rate": 1e-06,
"loss": 0.0007,
"step": 410
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 245.70983123779297,
"epoch": 68.41379310344827,
"grad_norm": 0.47997056233420055,
"kl": 0.0889892578125,
"learning_rate": 1e-06,
"loss": 0.0009,
"reward": 0.848214328289032,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.848214328289032,
"step": 411
},
{
"clip_ratio": 0.0004889596821158193,
"epoch": 68.55172413793103,
"grad_norm": 0.23762324351012565,
"kl": 0.1072998046875,
"learning_rate": 1e-06,
"loss": 0.0004,
"step": 412
},
{
"batch_accuracy": 0.9241071428571428,
"clip_ratio": 0.0,
"completion_length": 208.18304443359375,
"epoch": 68.6896551724138,
"grad_norm": 0.7607902677997762,
"kl": 0.1046142578125,
"learning_rate": 1e-06,
"loss": 0.0113,
"reward": 0.9241071939468384,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.9241071939468384,
"step": 413
},
{
"clip_ratio": 0.0,
"epoch": 68.82758620689656,
"grad_norm": 0.9318297670937001,
"kl": 0.112548828125,
"learning_rate": 1e-06,
"loss": 0.0106,
"step": 414
},
{
"batch_accuracy": 0.8392857142857143,
"clip_ratio": 0.0,
"completion_length": 248.25447845458984,
"epoch": 69.13793103448276,
"grad_norm": 0.41907478180367097,
"kl": 0.1656494140625,
"learning_rate": 1e-06,
"loss": 0.005,
"reward": 0.839285746216774,
"reward_std": 0.05050762742757797,
"rewards/unified_reward_func": 0.839285746216774,
"step": 415
},
{
"clip_ratio": 0.0014105399022810161,
"epoch": 69.27586206896552,
"grad_norm": 0.25769744775290754,
"kl": 0.183837890625,
"learning_rate": 1e-06,
"loss": 0.0043,
"step": 416
},
{
"batch_accuracy": 0.8125,
"clip_ratio": 0.0,
"completion_length": 237.3348388671875,
"epoch": 69.41379310344827,
"grad_norm": 0.189936523053462,
"kl": 0.169921875,
"learning_rate": 1e-06,
"loss": -0.0057,
"reward": 0.8125000298023224,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8125000298023224,
"step": 417
},
{
"clip_ratio": 0.00024228040274465457,
"epoch": 69.55172413793103,
"grad_norm": 0.14852661920422203,
"kl": 0.16748046875,
"learning_rate": 1e-06,
"loss": -0.0061,
"step": 418
},
{
"batch_accuracy": 0.8794642857142857,
"clip_ratio": 0.0,
"completion_length": 240.67412185668945,
"epoch": 69.6896551724138,
"grad_norm": 0.30067919176795704,
"kl": 0.1591796875,
"learning_rate": 1e-06,
"loss": 0.0019,
"reward": 0.8794643133878708,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.8794643133878708,
"step": 419
},
{
"clip_ratio": 0.0004018953113700263,
"epoch": 69.82758620689656,
"grad_norm": 0.19383722712493728,
"kl": 0.16796875,
"learning_rate": 1e-06,
"loss": 0.0015,
"step": 420
},
{
"batch_accuracy": 0.7678571428571428,
"clip_ratio": 0.0,
"completion_length": 224.6696548461914,
"epoch": 70.13793103448276,
"grad_norm": 0.6500836334692476,
"kl": 0.2109375,
"learning_rate": 1e-06,
"loss": -0.004,
"reward": 0.7678571790456772,
"reward_std": 0.05050762742757797,
"rewards/unified_reward_func": 0.7678571790456772,
"step": 421
},
{
"clip_ratio": 0.00046241712698247284,
"epoch": 70.27586206896552,
"grad_norm": 0.38904128273042526,
"kl": 0.204345703125,
"learning_rate": 1e-06,
"loss": -0.005,
"step": 422
},
{
"batch_accuracy": 0.9553571428571428,
"clip_ratio": 0.0,
"completion_length": 212.03125762939453,
"epoch": 70.41379310344827,
"grad_norm": 0.5739109252892366,
"kl": 0.4423828125,
"learning_rate": 1e-06,
"loss": -0.0116,
"reward": 0.9553571492433548,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9553571492433548,
"step": 423
},
{
"clip_ratio": 8.943354623625055e-05,
"epoch": 70.55172413793103,
"grad_norm": 0.16299775366551209,
"kl": 0.15478515625,
"learning_rate": 1e-06,
"loss": -0.012,
"step": 424
},
{
"batch_accuracy": 0.8794642857142857,
"clip_ratio": 0.0,
"completion_length": 228.59375762939453,
"epoch": 70.6896551724138,
"grad_norm": 0.38581004341045927,
"kl": 0.68408203125,
"learning_rate": 1e-06,
"loss": 0.0007,
"reward": 0.8794642984867096,
"reward_std": 0.018483899533748627,
"rewards/unified_reward_func": 0.8794642984867096,
"step": 425
},
{
"clip_ratio": 0.0001823043276090175,
"epoch": 70.82758620689656,
"grad_norm": 2.0528087697459636,
"kl": 0.314453125,
"learning_rate": 1e-06,
"loss": 0.0015,
"step": 426
},
{
"batch_accuracy": 0.8214285714285714,
"clip_ratio": 0.0,
"completion_length": 242.22768783569336,
"epoch": 71.13793103448276,
"grad_norm": 0.04625840730424008,
"kl": 0.23095703125,
"learning_rate": 1e-06,
"loss": 0.0002,
"reward": 0.8214286267757416,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.8214286267757416,
"step": 427
},
{
"clip_ratio": 0.0,
"epoch": 71.27586206896552,
"grad_norm": 0.03143840003558422,
"kl": 0.2021484375,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 428
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 233.1562614440918,
"epoch": 71.41379310344827,
"grad_norm": 0.5009023723263193,
"kl": 0.207763671875,
"learning_rate": 1e-06,
"loss": 0.003,
"reward": 0.8482143431901932,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8482143431901932,
"step": 429
},
{
"clip_ratio": 0.0003354004038556013,
"epoch": 71.55172413793103,
"grad_norm": 0.2538041516707458,
"kl": 0.1826171875,
"learning_rate": 1e-06,
"loss": 0.0022,
"step": 430
},
{
"batch_accuracy": 0.9151785714285714,
"clip_ratio": 0.0,
"completion_length": 222.32143783569336,
"epoch": 71.6896551724138,
"grad_norm": 0.6051502407175923,
"kl": 0.1630859375,
"learning_rate": 1e-06,
"loss": 0.0089,
"reward": 0.9151785969734192,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.9151785969734192,
"step": 431
},
{
"clip_ratio": 0.0003940899041481316,
"epoch": 71.82758620689656,
"grad_norm": 0.2518463742920773,
"kl": 0.155517578125,
"learning_rate": 1e-06,
"loss": 0.0084,
"step": 432
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 225.35268783569336,
"epoch": 72.13793103448276,
"grad_norm": 2.3905013205508494,
"kl": 0.99169921875,
"learning_rate": 1e-06,
"loss": 0.001,
"reward": 0.8482143431901932,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8482143431901932,
"step": 433
},
{
"clip_ratio": 0.0006973157869651914,
"epoch": 72.27586206896552,
"grad_norm": 0.23797546293977923,
"kl": 0.26318359375,
"learning_rate": 1e-06,
"loss": 0.0003,
"step": 434
},
{
"batch_accuracy": 0.84375,
"clip_ratio": 0.0,
"completion_length": 217.71875762939453,
"epoch": 72.41379310344827,
"grad_norm": 0.7831567380243376,
"kl": 0.1953125,
"learning_rate": 1e-06,
"loss": 0.002,
"reward": 0.8437500298023224,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.8437500298023224,
"step": 435
},
{
"clip_ratio": 0.0015309452719520777,
"epoch": 72.55172413793103,
"grad_norm": 0.34815002643262916,
"kl": 0.27490234375,
"learning_rate": 1e-06,
"loss": 0.0013,
"step": 436
},
{
"batch_accuracy": 0.9196428571428572,
"clip_ratio": 0.0,
"completion_length": 218.4241180419922,
"epoch": 72.6896551724138,
"grad_norm": 0.398082329113118,
"kl": 0.168701171875,
"learning_rate": 1e-06,
"loss": -0.0058,
"reward": 0.9196428656578064,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9196428656578064,
"step": 437
},
{
"clip_ratio": 0.00032450424623675644,
"epoch": 72.82758620689656,
"grad_norm": 0.20485368896653353,
"kl": 0.172119140625,
"learning_rate": 1e-06,
"loss": -0.0062,
"step": 438
},
{
"batch_accuracy": 0.8169642857142857,
"clip_ratio": 0.0,
"completion_length": 211.16965103149414,
"epoch": 73.13793103448276,
"grad_norm": 0.33955964196499566,
"kl": 0.20263671875,
"learning_rate": 1e-06,
"loss": 0.0007,
"reward": 0.8169643133878708,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.8169643133878708,
"step": 439
},
{
"clip_ratio": 2.829975164786447e-05,
"epoch": 73.27586206896552,
"grad_norm": 0.26818712559841884,
"kl": 0.270751953125,
"learning_rate": 1e-06,
"loss": 0.0003,
"step": 440
},
{
"batch_accuracy": 0.8794642857142857,
"clip_ratio": 0.0,
"completion_length": 245.52680206298828,
"epoch": 73.41379310344827,
"grad_norm": 0.6770300911622293,
"kl": 0.58935546875,
"learning_rate": 1e-06,
"loss": 0.005,
"reward": 0.8794643133878708,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.8794643133878708,
"step": 441
},
{
"clip_ratio": 0.00035833044967148453,
"epoch": 73.55172413793103,
"grad_norm": 79.74265910678565,
"kl": 35.379638671875,
"learning_rate": 1e-06,
"loss": 0.0388,
"step": 442
},
{
"batch_accuracy": 0.9241071428571428,
"clip_ratio": 0.0,
"completion_length": 220.05358123779297,
"epoch": 73.6896551724138,
"grad_norm": 1.2885073773349964,
"kl": 1.6875,
"learning_rate": 1e-06,
"loss": 0.0027,
"reward": 0.924107164144516,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.924107164144516,
"step": 443
},
{
"clip_ratio": 5.87406029808335e-05,
"epoch": 73.82758620689656,
"grad_norm": 0.23799297864282562,
"kl": 0.529541015625,
"learning_rate": 1e-06,
"loss": 0.0014,
"step": 444
},
{
"batch_accuracy": 0.9151785714285714,
"clip_ratio": 0.0,
"completion_length": 244.63394165039062,
"epoch": 74.13793103448276,
"grad_norm": 0.40842932428295264,
"kl": 0.1495361328125,
"learning_rate": 1e-06,
"loss": -0.0062,
"reward": 0.915178582072258,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.915178582072258,
"step": 445
},
{
"clip_ratio": 0.0008320850902236998,
"epoch": 74.27586206896552,
"grad_norm": 0.2382784575121656,
"kl": 0.1640625,
"learning_rate": 1e-06,
"loss": -0.0067,
"step": 446
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 230.9866180419922,
"epoch": 74.41379310344827,
"grad_norm": 0.1997408281341476,
"kl": 0.208740234375,
"learning_rate": 1e-06,
"loss": 0.0018,
"reward": 0.8482143133878708,
"reward_std": 0.016532503068447113,
"rewards/unified_reward_func": 0.8482143133878708,
"step": 447
},
{
"clip_ratio": 0.000311763898935169,
"epoch": 74.55172413793103,
"grad_norm": 0.1335107548690185,
"kl": 0.16796875,
"learning_rate": 1e-06,
"loss": 0.0015,
"step": 448
},
{
"batch_accuracy": 0.8928571428571428,
"clip_ratio": 0.0,
"completion_length": 222.02679824829102,
"epoch": 74.6896551724138,
"grad_norm": 0.10165897354603974,
"kl": 0.28369140625,
"learning_rate": 1e-06,
"loss": 0.0003,
"reward": 0.892857164144516,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.892857164144516,
"step": 449
},
{
"epoch": 74.82758620689656,
"grad_norm": 0.02722702905999937,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 450
},
{
"epoch": 74.82758620689656,
"eval_batch_accuracy": 0.7607142857142858,
"eval_clip_ratio": 0.0,
"eval_completion_length": 239.97046305338543,
"eval_kl": 0.4384765625,
"eval_loss": 0.010263421572744846,
"eval_reward": 0.760714324315389,
"eval_reward_std": 0.10253030310074489,
"eval_rewards/unified_reward_func": 0.760714324315389,
"eval_runtime": 292.1226,
"eval_samples_per_second": 0.342,
"eval_steps_per_second": 0.007,
"step": 450
},
{
"batch_accuracy": 0.7767857142857143,
"clip_ratio": 0.0,
"completion_length": 292.33483123779297,
"epoch": 75.13793103448276,
"grad_norm": 1.2280644592028591,
"kl": 0.53375244140625,
"learning_rate": 1e-06,
"loss": -0.0072,
"reward": 0.776785746216774,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.776785746216774,
"step": 451
},
{
"clip_ratio": 0.00022370762599166483,
"epoch": 75.27586206896552,
"grad_norm": 3294.426305780333,
"kl": 0.3575439453125,
"learning_rate": 1e-06,
"loss": 1.0297,
"step": 452
},
{
"batch_accuracy": 0.9642857142857143,
"clip_ratio": 0.0,
"completion_length": 210.72322463989258,
"epoch": 75.41379310344827,
"grad_norm": 0.012716440951391888,
"kl": 0.1534423828125,
"learning_rate": 1e-06,
"loss": 0.0002,
"reward": 0.9642857313156128,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.9642857313156128,
"step": 453
},
{
"clip_ratio": 0.0,
"epoch": 75.55172413793103,
"grad_norm": 0.012240949557013305,
"kl": 0.1512451171875,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 454
},
{
"batch_accuracy": 0.9196428571428571,
"clip_ratio": 0.0,
"completion_length": 209.94643783569336,
"epoch": 75.6896551724138,
"grad_norm": 0.5247671039188904,
"kl": 0.160400390625,
"learning_rate": 1e-06,
"loss": 0.0076,
"reward": 0.9196429252624512,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9196429252624512,
"step": 455
},
{
"clip_ratio": 0.00023868290008977056,
"epoch": 75.82758620689656,
"grad_norm": 0.2792572669606453,
"kl": 0.161865234375,
"learning_rate": 1e-06,
"loss": 0.0069,
"step": 456
},
{
"batch_accuracy": 0.8571428571428572,
"clip_ratio": 0.0,
"completion_length": 188.30804443359375,
"epoch": 76.13793103448276,
"grad_norm": 0.04946528678616401,
"kl": 0.218017578125,
"learning_rate": 1e-06,
"loss": 0.0002,
"reward": 0.8571428656578064,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.8571428656578064,
"step": 457
},
{
"clip_ratio": 0.0,
"epoch": 76.27586206896552,
"grad_norm": 0.039462119096367015,
"kl": 0.21142578125,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 458
},
{
"batch_accuracy": 0.8928571428571428,
"clip_ratio": 0.0,
"completion_length": 229.00893783569336,
"epoch": 76.41379310344827,
"grad_norm": 0.5075704410401828,
"kl": 0.501220703125,
"learning_rate": 1e-06,
"loss": 0.0005,
"reward": 0.8928571939468384,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.8928571939468384,
"step": 459
},
{
"clip_ratio": 0.0,
"epoch": 76.55172413793103,
"grad_norm": 0.03775215604367821,
"kl": 0.20849609375,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 460
},
{
"batch_accuracy": 0.8616071428571428,
"clip_ratio": 0.0,
"completion_length": 254.00000762939453,
"epoch": 76.6896551724138,
"grad_norm": 0.7803983058510199,
"kl": 0.3798828125,
"learning_rate": 1e-06,
"loss": 0.0166,
"reward": 0.861607164144516,
"reward_std": 0.07966703735291958,
"rewards/unified_reward_func": 0.861607164144516,
"step": 461
},
{
"clip_ratio": 0.0014210399895091541,
"epoch": 76.82758620689656,
"grad_norm": 0.49834308317287584,
"kl": 0.298583984375,
"learning_rate": 1e-06,
"loss": 0.0151,
"step": 462
},
{
"batch_accuracy": 0.8392857142857143,
"clip_ratio": 0.0,
"completion_length": 242.5759048461914,
"epoch": 77.13793103448276,
"grad_norm": 0.5803647637709338,
"kl": 0.32666015625,
"learning_rate": 1e-06,
"loss": -0.0033,
"reward": 0.8392857313156128,
"reward_std": 0.05050762742757797,
"rewards/unified_reward_func": 0.8392857313156128,
"step": 463
},
{
"clip_ratio": 0.0003732922159542795,
"epoch": 77.27586206896552,
"grad_norm": 0.37156821061329787,
"kl": 0.266357421875,
"learning_rate": 1e-06,
"loss": -0.0041,
"step": 464
},
{
"batch_accuracy": 0.8392857142857142,
"clip_ratio": 0.0,
"completion_length": 227.93304443359375,
"epoch": 77.41379310344827,
"grad_norm": 0.5552624609943606,
"kl": 0.172119140625,
"learning_rate": 1e-06,
"loss": -0.0093,
"reward": 0.8392857611179352,
"reward_std": 0.05050762742757797,
"rewards/unified_reward_func": 0.8392857611179352,
"step": 465
},
{
"clip_ratio": 0.0005572987865889445,
"epoch": 77.55172413793103,
"grad_norm": 0.29331584969110686,
"kl": 0.181640625,
"learning_rate": 1e-06,
"loss": -0.0101,
"step": 466
},
{
"batch_accuracy": 0.9508928571428571,
"clip_ratio": 0.0,
"completion_length": 206.97768783569336,
"epoch": 77.6896551724138,
"grad_norm": 0.4342585602914251,
"kl": 0.2197265625,
"learning_rate": 1e-06,
"loss": 0.0008,
"reward": 0.9508928954601288,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.9508928954601288,
"step": 467
},
{
"clip_ratio": 0.0007359530427493155,
"epoch": 77.82758620689656,
"grad_norm": 0.9475669204813898,
"kl": 0.556640625,
"learning_rate": 1e-06,
"loss": 0.0006,
"step": 468
},
{
"batch_accuracy": 0.7723214285714286,
"clip_ratio": 0.0,
"completion_length": 252.02233123779297,
"epoch": 78.13793103448276,
"grad_norm": 0.6308618189250425,
"kl": 0.570556640625,
"learning_rate": 1e-06,
"loss": 0.0042,
"reward": 0.7723214626312256,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.7723214626312256,
"step": 469
},
{
"clip_ratio": 0.0002528606928535737,
"epoch": 78.27586206896552,
"grad_norm": 0.3335563972152877,
"kl": 0.587158203125,
"learning_rate": 1e-06,
"loss": 0.0032,
"step": 470
},
{
"batch_accuracy": 0.9196428571428572,
"clip_ratio": 0.0,
"completion_length": 234.14286041259766,
"epoch": 78.41379310344827,
"grad_norm": 0.4791093710644942,
"kl": 0.271728515625,
"learning_rate": 1e-06,
"loss": 0.0074,
"reward": 0.9196428656578064,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9196428656578064,
"step": 471
},
{
"clip_ratio": 0.00046849474165355787,
"epoch": 78.55172413793103,
"grad_norm": 0.28861887009948634,
"kl": 0.249267578125,
"learning_rate": 1e-06,
"loss": 0.0068,
"step": 472
},
{
"batch_accuracy": 0.9285714285714286,
"clip_ratio": 0.0,
"completion_length": 206.95983123779297,
"epoch": 78.6896551724138,
"grad_norm": 0.13724663375532223,
"kl": 0.2587890625,
"learning_rate": 1e-06,
"loss": 0.0003,
"reward": 0.9285714328289032,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.9285714328289032,
"step": 473
},
{
"clip_ratio": 0.0,
"epoch": 78.82758620689656,
"grad_norm": 0.037068066277568514,
"kl": 0.216064453125,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 474
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 199.35715103149414,
"epoch": 79.13793103448276,
"grad_norm": 0.3264825065714323,
"kl": 0.2763671875,
"learning_rate": 1e-06,
"loss": -0.009,
"reward": 0.848214328289032,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.848214328289032,
"step": 475
},
{
"clip_ratio": 0.0003418240521568805,
"epoch": 79.27586206896552,
"grad_norm": 0.18676471981083279,
"kl": 0.2568359375,
"learning_rate": 1e-06,
"loss": -0.0093,
"step": 476
},
{
"batch_accuracy": 0.8883928571428571,
"clip_ratio": 0.0,
"completion_length": 233.7946548461914,
"epoch": 79.41379310344827,
"grad_norm": 1.101668515839696,
"kl": 0.56103515625,
"learning_rate": 1e-06,
"loss": -0.0019,
"reward": 0.8883928954601288,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.8883928954601288,
"step": 477
},
{
"clip_ratio": 0.00017312313138972968,
"epoch": 79.55172413793103,
"grad_norm": 0.1583937351723072,
"kl": 0.229248046875,
"learning_rate": 1e-06,
"loss": -0.0022,
"step": 478
},
{
"batch_accuracy": 0.8035714285714286,
"clip_ratio": 0.0,
"completion_length": 217.97322463989258,
"epoch": 79.6896551724138,
"grad_norm": 0.9780994951787898,
"kl": 0.823974609375,
"learning_rate": 1e-06,
"loss": 0.0107,
"reward": 0.8035714626312256,
"reward_std": 0.05050762742757797,
"rewards/unified_reward_func": 0.8035714626312256,
"step": 479
},
{
"clip_ratio": 0.0002679546087165363,
"epoch": 79.82758620689656,
"grad_norm": 0.4882068912890497,
"kl": 0.30224609375,
"learning_rate": 1e-06,
"loss": 0.0094,
"step": 480
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 234.30805206298828,
"epoch": 80.13793103448276,
"grad_norm": 0.25335686041912453,
"kl": 0.414306640625,
"learning_rate": 1e-06,
"loss": -0.0248,
"reward": 0.8839285969734192,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839285969734192,
"step": 481
},
{
"clip_ratio": 8.276415883301524e-05,
"epoch": 80.27586206896552,
"grad_norm": 0.11536539576915139,
"kl": 0.22021484375,
"learning_rate": 1e-06,
"loss": -0.0252,
"step": 482
},
{
"batch_accuracy": 0.7857142857142857,
"clip_ratio": 0.0,
"completion_length": 209.30804061889648,
"epoch": 80.41379310344827,
"grad_norm": 0.11998413938965553,
"kl": 0.2314453125,
"learning_rate": 1e-06,
"loss": 0.0002,
"reward": 0.785714328289032,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.785714328289032,
"step": 483
},
{
"clip_ratio": 0.0,
"epoch": 80.55172413793103,
"grad_norm": 0.07272808970933248,
"kl": 0.202392578125,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 484
},
{
"batch_accuracy": 0.9151785714285714,
"clip_ratio": 0.0,
"completion_length": 220.6696548461914,
"epoch": 80.6896551724138,
"grad_norm": 0.20330006705789813,
"kl": 0.232421875,
"learning_rate": 1e-06,
"loss": -0.0026,
"reward": 0.9151786118745804,
"reward_std": 0.018483899533748627,
"rewards/unified_reward_func": 0.9151786118745804,
"step": 485
},
{
"clip_ratio": 8.486562728649005e-05,
"epoch": 80.82758620689656,
"grad_norm": 0.13781009873072228,
"kl": 0.181640625,
"learning_rate": 1e-06,
"loss": -0.0029,
"step": 486
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 207.99108123779297,
"epoch": 81.13793103448276,
"grad_norm": 0.4285326309410844,
"kl": 0.1611328125,
"learning_rate": 1e-06,
"loss": 0.0008,
"reward": 0.8839286267757416,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839286267757416,
"step": 487
},
{
"clip_ratio": 0.0002343350297451252,
"epoch": 81.27586206896552,
"grad_norm": 0.20850823605286758,
"kl": 0.1688232421875,
"learning_rate": 1e-06,
"loss": 0.0004,
"step": 488
},
{
"batch_accuracy": 0.9241071428571428,
"clip_ratio": 0.0,
"completion_length": 216.19644165039062,
"epoch": 81.41379310344827,
"grad_norm": 0.30122462050916987,
"kl": 0.1904296875,
"learning_rate": 1e-06,
"loss": -0.0023,
"reward": 0.924107164144516,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.924107164144516,
"step": 489
},
{
"clip_ratio": 8.710801193956286e-05,
"epoch": 81.55172413793103,
"grad_norm": 1.4111707805241815,
"kl": 0.67236328125,
"learning_rate": 1e-06,
"loss": -0.0021,
"step": 490
},
{
"batch_accuracy": 0.7678571428571428,
"clip_ratio": 0.0,
"completion_length": 244.08037185668945,
"epoch": 81.6896551724138,
"grad_norm": 0.4743000711442227,
"kl": 0.30810546875,
"learning_rate": 1e-06,
"loss": -0.0068,
"reward": 0.7678571790456772,
"reward_std": 0.033065006136894226,
"rewards/unified_reward_func": 0.7678571790456772,
"step": 491
},
{
"clip_ratio": 0.00022754022211302072,
"epoch": 81.82758620689656,
"grad_norm": 0.26704613852928916,
"kl": 0.24658203125,
"learning_rate": 1e-06,
"loss": -0.0074,
"step": 492
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 246.3928680419922,
"epoch": 82.13793103448276,
"grad_norm": 0.5254256464673122,
"kl": 0.374267578125,
"learning_rate": 1e-06,
"loss": -0.0061,
"reward": 0.8839285969734192,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839285969734192,
"step": 493
},
{
"clip_ratio": 0.0004009578133263858,
"epoch": 82.27586206896552,
"grad_norm": 0.2680176309600896,
"kl": 0.301513671875,
"learning_rate": 1e-06,
"loss": -0.0067,
"step": 494
},
{
"batch_accuracy": 0.8928571428571428,
"clip_ratio": 0.0,
"completion_length": 214.97768783569336,
"epoch": 82.41379310344827,
"grad_norm": 0.7978444005747602,
"kl": 0.912353515625,
"learning_rate": 1e-06,
"loss": 0.0009,
"reward": 0.8928571939468384,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.8928571939468384,
"step": 495
},
{
"clip_ratio": 0.0,
"epoch": 82.55172413793103,
"grad_norm": 0.025686755334383627,
"kl": 0.19580078125,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 496
},
{
"batch_accuracy": 0.8169642857142857,
"clip_ratio": 0.0,
"completion_length": 236.85269165039062,
"epoch": 82.6896551724138,
"grad_norm": 0.11684909474736647,
"kl": 0.25927734375,
"learning_rate": 1e-06,
"loss": 0.0007,
"reward": 0.816964328289032,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.816964328289032,
"step": 497
},
{
"clip_ratio": 3.523856503306888e-05,
"epoch": 82.82758620689656,
"grad_norm": 0.06796555451043418,
"kl": 0.19921875,
"learning_rate": 1e-06,
"loss": 0.0005,
"step": 498
},
{
"batch_accuracy": 0.7276785714285714,
"clip_ratio": 0.0,
"completion_length": 245.62054443359375,
"epoch": 83.13793103448276,
"grad_norm": 0.8222045042128722,
"kl": 0.2041015625,
"learning_rate": 1e-06,
"loss": 0.0066,
"reward": 0.7276785969734192,
"reward_std": 0.06313453428447247,
"rewards/unified_reward_func": 0.7276785969734192,
"step": 499
},
{
"epoch": 83.27586206896552,
"grad_norm": 0.5300778222275132,
"learning_rate": 1e-06,
"loss": 0.0054,
"step": 500
},
{
"epoch": 83.27586206896552,
"eval_batch_accuracy": 0.7404761904761905,
"eval_clip_ratio": 0.0,
"eval_completion_length": 242.2316131591797,
"eval_kl": 0.205615234375,
"eval_loss": -0.00046114774886518717,
"eval_reward": 0.7404762228329976,
"eval_reward_std": 0.11178621302048365,
"eval_rewards/unified_reward_func": 0.7404762228329976,
"eval_runtime": 284.5712,
"eval_samples_per_second": 0.351,
"eval_steps_per_second": 0.007,
"step": 500
},
{
"batch_accuracy": 0.9553571428571428,
"clip_ratio": 0.0003930836610379629,
"completion_length": 223.55358123779297,
"epoch": 83.41379310344827,
"grad_norm": 0.3338264355322204,
"kl": 0.15478515625,
"learning_rate": 1e-06,
"loss": -0.0017,
"reward": 0.9553571939468384,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9553571939468384,
"step": 501
},
{
"clip_ratio": 0.0004324326873756945,
"epoch": 83.55172413793103,
"grad_norm": 0.18037033885748482,
"kl": 0.1173095703125,
"learning_rate": 1e-06,
"loss": -0.0021,
"step": 502
},
{
"batch_accuracy": 0.8883928571428571,
"clip_ratio": 0.0,
"completion_length": 233.3437614440918,
"epoch": 83.6896551724138,
"grad_norm": 0.24428580583668974,
"kl": 0.1396484375,
"learning_rate": 1e-06,
"loss": -0.0004,
"reward": 0.8883928805589676,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.8883928805589676,
"step": 503
},
{
"clip_ratio": 0.00011091393389506266,
"epoch": 83.82758620689656,
"grad_norm": 0.12753916151633782,
"kl": 0.1373291015625,
"learning_rate": 1e-06,
"loss": -0.0006,
"step": 504
},
{
"batch_accuracy": 0.8928571428571428,
"clip_ratio": 0.0,
"completion_length": 217.25447463989258,
"epoch": 84.13793103448276,
"grad_norm": 0.03625836028576517,
"kl": 0.1346435546875,
"learning_rate": 1e-06,
"loss": 0.0001,
"reward": 0.892857164144516,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.892857164144516,
"step": 505
},
{
"clip_ratio": 0.0,
"epoch": 84.27586206896552,
"grad_norm": 0.022544787709227847,
"kl": 0.109130859375,
"learning_rate": 1e-06,
"loss": 0.0001,
"step": 506
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 238.36161422729492,
"epoch": 84.41379310344827,
"grad_norm": 0.7982539820094933,
"kl": 0.2203369140625,
"learning_rate": 1e-06,
"loss": -0.0081,
"reward": 0.8482143431901932,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8482143431901932,
"step": 507
},
{
"clip_ratio": 0.00042628360097296536,
"epoch": 84.55172413793103,
"grad_norm": 0.3511699215249297,
"kl": 0.2071533203125,
"learning_rate": 1e-06,
"loss": -0.0089,
"step": 508
},
{
"batch_accuracy": 0.8928571428571428,
"clip_ratio": 0.0,
"completion_length": 260.40625381469727,
"epoch": 84.6896551724138,
"grad_norm": 0.011861750126758629,
"kl": 0.08392333984375,
"learning_rate": 1e-06,
"loss": 0.0001,
"reward": 0.8928571939468384,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.8928571939468384,
"step": 509
},
{
"clip_ratio": 0.0,
"epoch": 84.82758620689656,
"grad_norm": 0.0115087349123355,
"kl": 0.08526611328125,
"learning_rate": 1e-06,
"loss": 0.0001,
"step": 510
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 214.8973274230957,
"epoch": 85.13793103448276,
"grad_norm": 0.36929124899796995,
"kl": 0.1583251953125,
"learning_rate": 1e-06,
"loss": -0.0044,
"reward": 0.8839286118745804,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839286118745804,
"step": 511
},
{
"clip_ratio": 0.00018426612950861454,
"epoch": 85.27586206896552,
"grad_norm": 0.15182488091269972,
"kl": 0.1295166015625,
"learning_rate": 1e-06,
"loss": -0.0049,
"step": 512
},
{
"batch_accuracy": 0.9508928571428572,
"clip_ratio": 0.0,
"completion_length": 234.2321548461914,
"epoch": 85.41379310344827,
"grad_norm": 0.3553720690267824,
"kl": 0.091796875,
"learning_rate": 1e-06,
"loss": -0.0045,
"reward": 0.9508928656578064,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.9508928656578064,
"step": 513
},
{
"clip_ratio": 0.00010698674213927006,
"epoch": 85.55172413793103,
"grad_norm": 0.21153475186501686,
"kl": 0.114990234375,
"learning_rate": 1e-06,
"loss": -0.005,
"step": 514
},
{
"batch_accuracy": 0.8794642857142857,
"clip_ratio": 0.0,
"completion_length": 261.16965103149414,
"epoch": 85.6896551724138,
"grad_norm": 0.4148941424286881,
"kl": 0.1607666015625,
"learning_rate": 1e-06,
"loss": -0.0028,
"reward": 0.879464328289032,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.879464328289032,
"step": 515
},
{
"clip_ratio": 0.0002979067139676772,
"epoch": 85.82758620689656,
"grad_norm": 0.2795928071224825,
"kl": 0.1551513671875,
"learning_rate": 1e-06,
"loss": -0.0035,
"step": 516
},
{
"batch_accuracy": 0.84375,
"clip_ratio": 0.0,
"completion_length": 253.6428680419922,
"epoch": 86.13793103448276,
"grad_norm": 0.27804716865753637,
"kl": 0.17291259765625,
"learning_rate": 1e-06,
"loss": -0.0075,
"reward": 0.8437500447034836,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.8437500447034836,
"step": 517
},
{
"clip_ratio": 0.0001618318710825406,
"epoch": 86.27586206896552,
"grad_norm": 0.20800322960868664,
"kl": 0.203125,
"learning_rate": 1e-06,
"loss": -0.0078,
"step": 518
},
{
"batch_accuracy": 0.9285714285714286,
"clip_ratio": 0.0,
"completion_length": 245.8303680419922,
"epoch": 86.41379310344827,
"grad_norm": 0.07559120994905119,
"kl": 0.1693115234375,
"learning_rate": 1e-06,
"loss": 0.0002,
"reward": 0.9285714328289032,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.9285714328289032,
"step": 519
},
{
"clip_ratio": 0.0,
"epoch": 86.55172413793103,
"grad_norm": 0.027510075620572266,
"kl": 0.1263427734375,
"learning_rate": 1e-06,
"loss": 0.0001,
"step": 520
},
{
"batch_accuracy": 0.8080357142857143,
"clip_ratio": 0.0,
"completion_length": 257.6651954650879,
"epoch": 86.6896551724138,
"grad_norm": 0.317410842856532,
"kl": 0.18505859375,
"learning_rate": 1e-06,
"loss": -0.0014,
"reward": 0.808035746216774,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.808035746216774,
"step": 521
},
{
"clip_ratio": 0.00045395906636258587,
"epoch": 86.82758620689656,
"grad_norm": 0.1853671737450051,
"kl": 0.19140625,
"learning_rate": 1e-06,
"loss": -0.0018,
"step": 522
},
{
"batch_accuracy": 0.8214285714285714,
"clip_ratio": 0.0,
"completion_length": 237.3259048461914,
"epoch": 87.13793103448276,
"grad_norm": 0.0430564348245684,
"kl": 0.2119140625,
"learning_rate": 1e-06,
"loss": 0.0002,
"reward": 0.821428582072258,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.821428582072258,
"step": 523
},
{
"clip_ratio": 0.0,
"epoch": 87.27586206896552,
"grad_norm": 0.034899739644111305,
"kl": 0.1956787109375,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 524
},
{
"batch_accuracy": 0.8035714285714286,
"clip_ratio": 0.0,
"completion_length": 268.3705520629883,
"epoch": 87.41379310344827,
"grad_norm": 0.4141970723107917,
"kl": 0.1942138671875,
"learning_rate": 1e-06,
"loss": -0.0078,
"reward": 0.8035714626312256,
"reward_std": 0.05050762742757797,
"rewards/unified_reward_func": 0.8035714626312256,
"step": 525
},
{
"clip_ratio": 0.0006828630139352754,
"epoch": 87.55172413793103,
"grad_norm": 0.32788510076339705,
"kl": 0.2764892578125,
"learning_rate": 1e-06,
"loss": -0.0084,
"step": 526
},
{
"batch_accuracy": 0.9553571428571428,
"clip_ratio": 0.0,
"completion_length": 250.99108505249023,
"epoch": 87.6896551724138,
"grad_norm": 0.22739404521765016,
"kl": 0.126953125,
"learning_rate": 1e-06,
"loss": -0.004,
"reward": 0.9553571939468384,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9553571939468384,
"step": 527
},
{
"clip_ratio": 0.00020626326295314357,
"epoch": 87.82758620689656,
"grad_norm": 0.14809417054594567,
"kl": 0.1229248046875,
"learning_rate": 1e-06,
"loss": -0.0043,
"step": 528
},
{
"batch_accuracy": 0.8482142857142857,
"clip_ratio": 0.0,
"completion_length": 286.08484268188477,
"epoch": 88.13793103448276,
"grad_norm": 0.36924668148503553,
"kl": 0.1690673828125,
"learning_rate": 1e-06,
"loss": -0.0093,
"reward": 0.8482142984867096,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8482142984867096,
"step": 529
},
{
"clip_ratio": 0.0002911818137363298,
"epoch": 88.27586206896552,
"grad_norm": 0.24608301664562246,
"kl": 0.1900634765625,
"learning_rate": 1e-06,
"loss": -0.0098,
"step": 530
},
{
"batch_accuracy": 0.8571428571428572,
"clip_ratio": 0.0,
"completion_length": 227.38393783569336,
"epoch": 88.41379310344827,
"grad_norm": 0.3658054197303534,
"kl": 0.669677734375,
"learning_rate": 1e-06,
"loss": 0.0007,
"reward": 0.8571428656578064,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.8571428656578064,
"step": 531
},
{
"clip_ratio": 0.0,
"epoch": 88.55172413793103,
"grad_norm": 0.07118533950169958,
"kl": 0.24267578125,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 532
},
{
"batch_accuracy": 0.84375,
"clip_ratio": 0.0,
"completion_length": 245.21430206298828,
"epoch": 88.6896551724138,
"grad_norm": 0.4255380426635142,
"kl": 0.1558837890625,
"learning_rate": 1e-06,
"loss": -0.0099,
"reward": 0.8437500298023224,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.8437500298023224,
"step": 533
},
{
"clip_ratio": 0.0009154944564215839,
"epoch": 88.82758620689656,
"grad_norm": 0.25713411225878763,
"kl": 0.162841796875,
"learning_rate": 1e-06,
"loss": -0.0106,
"step": 534
},
{
"batch_accuracy": 0.8883928571428571,
"clip_ratio": 0.0,
"completion_length": 273.3437614440918,
"epoch": 89.13793103448276,
"grad_norm": 0.24874726271726094,
"kl": 0.154541015625,
"learning_rate": 1e-06,
"loss": 0.0008,
"reward": 0.8883928954601288,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.8883928954601288,
"step": 535
},
{
"clip_ratio": 0.000315327662974596,
"epoch": 89.27586206896552,
"grad_norm": 0.15339425804522938,
"kl": 0.2099609375,
"learning_rate": 1e-06,
"loss": 0.0007,
"step": 536
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 223.38394165039062,
"epoch": 89.41379310344827,
"grad_norm": 0.43792988145907863,
"kl": 0.3629150390625,
"learning_rate": 1e-06,
"loss": -0.0038,
"reward": 0.8839286267757416,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839286267757416,
"step": 537
},
{
"clip_ratio": 0.0004368049849290401,
"epoch": 89.55172413793103,
"grad_norm": 0.22254866311488028,
"kl": 0.355224609375,
"learning_rate": 1e-06,
"loss": -0.0042,
"step": 538
},
{
"batch_accuracy": 0.7723214285714286,
"clip_ratio": 0.0,
"completion_length": 285.77233123779297,
"epoch": 89.6896551724138,
"grad_norm": 0.5234454954414901,
"kl": 0.437255859375,
"learning_rate": 1e-06,
"loss": 0.003,
"reward": 0.7723214477300644,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.7723214477300644,
"step": 539
},
{
"clip_ratio": 0.00027410148322815076,
"epoch": 89.82758620689656,
"grad_norm": 0.2959107381794968,
"kl": 0.2034912109375,
"learning_rate": 1e-06,
"loss": 0.0023,
"step": 540
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 258.15626525878906,
"epoch": 90.13793103448276,
"grad_norm": 0.5606663435729826,
"kl": 0.2216796875,
"learning_rate": 1e-06,
"loss": -0.0038,
"reward": 0.8839286118745804,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839286118745804,
"step": 541
},
{
"clip_ratio": 0.0004936900659231469,
"epoch": 90.27586206896552,
"grad_norm": 0.30058337241291133,
"kl": 0.228515625,
"learning_rate": 1e-06,
"loss": -0.0045,
"step": 542
},
{
"batch_accuracy": 1.0,
"clip_ratio": 0.0,
"completion_length": 230.7500114440918,
"epoch": 90.41379310344827,
"grad_norm": 0.013261529730841225,
"kl": 0.1427001953125,
"learning_rate": 1e-06,
"loss": 0.0001,
"reward": 1.0,
"reward_std": 0.0,
"rewards/unified_reward_func": 1.0,
"step": 543
},
{
"clip_ratio": 0.0,
"epoch": 90.55172413793103,
"grad_norm": 0.014812055834447134,
"kl": 0.146484375,
"learning_rate": 1e-06,
"loss": 0.0001,
"step": 544
},
{
"batch_accuracy": 0.8571428571428571,
"clip_ratio": 0.0,
"completion_length": 273.43304443359375,
"epoch": 90.6896551724138,
"grad_norm": 0.020177764439694507,
"kl": 0.1837158203125,
"learning_rate": 1e-06,
"loss": 0.0002,
"reward": 0.8571428805589676,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.8571428805589676,
"step": 545
},
{
"clip_ratio": 0.0,
"epoch": 90.82758620689656,
"grad_norm": 0.020471063976257625,
"kl": 0.184814453125,
"learning_rate": 1e-06,
"loss": 0.0002,
"step": 546
},
{
"batch_accuracy": 0.8169642857142857,
"clip_ratio": 0.0,
"completion_length": 289.08929443359375,
"epoch": 91.13793103448276,
"grad_norm": 0.3091335417653999,
"kl": 0.270751953125,
"learning_rate": 1e-06,
"loss": 0.0053,
"reward": 0.816964328289032,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.816964328289032,
"step": 547
},
{
"clip_ratio": 2.406623025308363e-05,
"epoch": 91.27586206896552,
"grad_norm": 0.1700906815409899,
"kl": 0.221435546875,
"learning_rate": 1e-06,
"loss": 0.0048,
"step": 548
},
{
"batch_accuracy": 0.9285714285714286,
"clip_ratio": 0.0,
"completion_length": 278.3571586608887,
"epoch": 91.41379310344827,
"grad_norm": 0.091506341158689,
"kl": 0.171630859375,
"learning_rate": 1e-06,
"loss": 0.0002,
"reward": 0.9285714626312256,
"reward_std": 0.0,
"rewards/unified_reward_func": 0.9285714626312256,
"step": 549
},
{
"epoch": 91.55172413793103,
"grad_norm": 0.014377936909430832,
"learning_rate": 1e-06,
"loss": 0.0001,
"step": 550
},
{
"epoch": 91.55172413793103,
"eval_batch_accuracy": 0.75,
"eval_clip_ratio": 0.0,
"eval_completion_length": 273.0262034098307,
"eval_kl": 0.3885416666666667,
"eval_loss": 0.011643623933196068,
"eval_reward": 0.7500000437100728,
"eval_reward_std": 0.12268384645382563,
"eval_rewards/unified_reward_func": 0.7500000437100728,
"eval_runtime": 301.6826,
"eval_samples_per_second": 0.331,
"eval_steps_per_second": 0.007,
"step": 550
},
{
"batch_accuracy": 0.8794642857142857,
"clip_ratio": 0.0,
"completion_length": 254.08929824829102,
"epoch": 91.6896551724138,
"grad_norm": 0.23783649200488552,
"kl": 0.18267822265625,
"learning_rate": 1e-06,
"loss": -0.0033,
"reward": 0.8794643431901932,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.8794643431901932,
"step": 551
},
{
"clip_ratio": 0.00017913367628352717,
"epoch": 91.82758620689656,
"grad_norm": 0.15644414933150733,
"kl": 0.185791015625,
"learning_rate": 1e-06,
"loss": -0.0036,
"step": 552
},
{
"batch_accuracy": 0.8303571428571428,
"clip_ratio": 0.0,
"completion_length": 251.42411422729492,
"epoch": 92.13793103448276,
"grad_norm": 0.42449283852885444,
"kl": 0.187744140625,
"learning_rate": 1e-06,
"loss": 0.0016,
"reward": 0.8303571790456772,
"reward_std": 0.04764331132173538,
"rewards/unified_reward_func": 0.8303571790456772,
"step": 553
},
{
"clip_ratio": 0.0004540284280665219,
"epoch": 92.27586206896552,
"grad_norm": 0.28459956459838937,
"kl": 0.18994140625,
"learning_rate": 1e-06,
"loss": 0.0007,
"step": 554
},
{
"batch_accuracy": 0.8794642857142857,
"clip_ratio": 0.0,
"completion_length": 269.4062614440918,
"epoch": 92.41379310344827,
"grad_norm": 0.519460289294838,
"kl": 0.41162109375,
"learning_rate": 1e-06,
"loss": -0.003,
"reward": 0.879464328289032,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.879464328289032,
"step": 555
},
{
"clip_ratio": 0.0007315961556741968,
"epoch": 92.55172413793103,
"grad_norm": 0.30233640375635246,
"kl": 0.505859375,
"learning_rate": 1e-06,
"loss": -0.0036,
"step": 556
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 263.8750114440918,
"epoch": 92.6896551724138,
"grad_norm": 0.5561022662042301,
"kl": 0.641845703125,
"learning_rate": 1e-06,
"loss": -0.0024,
"reward": 0.8839286118745804,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839286118745804,
"step": 557
},
{
"clip_ratio": 0.0005278162134345621,
"epoch": 92.82758620689656,
"grad_norm": 0.23851011229091273,
"kl": 0.26513671875,
"learning_rate": 1e-06,
"loss": -0.003,
"step": 558
},
{
"batch_accuracy": 0.7767857142857142,
"clip_ratio": 0.0,
"completion_length": 270.8884086608887,
"epoch": 93.13793103448276,
"grad_norm": 0.4536736642998327,
"kl": 0.234375,
"learning_rate": 1e-06,
"loss": 0.007,
"reward": 0.7767857685685158,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.7767857685685158,
"step": 559
},
{
"clip_ratio": 0.000577432889258489,
"epoch": 93.27586206896552,
"grad_norm": 0.285893962946527,
"kl": 0.244140625,
"learning_rate": 1e-06,
"loss": 0.0061,
"step": 560
},
{
"batch_accuracy": 0.9910714285714286,
"clip_ratio": 0.0,
"completion_length": 201.28125762939453,
"epoch": 93.41379310344827,
"grad_norm": 0.41331881846906104,
"kl": 0.231689453125,
"learning_rate": 1e-06,
"loss": 0.002,
"reward": 0.9910714626312256,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9910714626312256,
"step": 561
},
{
"clip_ratio": 0.0003733747871592641,
"epoch": 93.55172413793103,
"grad_norm": 0.25037100349015595,
"kl": 0.23583984375,
"learning_rate": 1e-06,
"loss": 0.0013,
"step": 562
},
{
"batch_accuracy": 0.9017857142857143,
"clip_ratio": 0.0,
"completion_length": 287.0625114440918,
"epoch": 93.6896551724138,
"grad_norm": 0.23791341082022063,
"kl": 0.342529296875,
"learning_rate": 1e-06,
"loss": -0.002,
"reward": 0.901785746216774,
"reward_std": 0.04764330945909023,
"rewards/unified_reward_func": 0.901785746216774,
"step": 563
},
{
"clip_ratio": 0.00034440189483575523,
"epoch": 93.82758620689656,
"grad_norm": 0.17440834096908142,
"kl": 0.340576171875,
"learning_rate": 1e-06,
"loss": -0.0025,
"step": 564
},
{
"batch_accuracy": 0.9151785714285714,
"clip_ratio": 0.0,
"completion_length": 243.61608505249023,
"epoch": 94.13793103448276,
"grad_norm": 0.6497367763504345,
"kl": 0.2451171875,
"learning_rate": 1e-06,
"loss": 0.0085,
"reward": 0.915178582072258,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.915178582072258,
"step": 565
},
{
"clip_ratio": 0.000318015729135368,
"epoch": 94.27586206896552,
"grad_norm": 0.7711925543158148,
"kl": 0.57373046875,
"learning_rate": 1e-06,
"loss": 0.0078,
"step": 566
},
{
"batch_accuracy": 0.9464285714285714,
"clip_ratio": 0.0,
"completion_length": 259.46429443359375,
"epoch": 94.41379310344827,
"grad_norm": 0.3528493235209404,
"kl": 0.35888671875,
"learning_rate": 1e-06,
"loss": 0.0024,
"reward": 0.9464286118745804,
"reward_std": 0.03111080639064312,
"rewards/unified_reward_func": 0.9464286118745804,
"step": 567
},
{
"clip_ratio": 0.0003265242121415213,
"epoch": 94.55172413793103,
"grad_norm": 0.23138814922762713,
"kl": 0.36865234375,
"learning_rate": 1e-06,
"loss": 0.0017,
"step": 568
},
{
"batch_accuracy": 0.7455357142857143,
"clip_ratio": 0.0,
"completion_length": 253.51340866088867,
"epoch": 94.6896551724138,
"grad_norm": 0.2436815317174032,
"kl": 0.54248046875,
"learning_rate": 1e-06,
"loss": -0.0038,
"reward": 0.7455357611179352,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.7455357611179352,
"step": 569
},
{
"clip_ratio": 0.0002118285046890378,
"epoch": 94.82758620689656,
"grad_norm": 0.19930797505222894,
"kl": 0.59716796875,
"learning_rate": 1e-06,
"loss": -0.0041,
"step": 570
},
{
"batch_accuracy": 0.84375,
"clip_ratio": 0.0,
"completion_length": 277.7232246398926,
"epoch": 95.13793103448276,
"grad_norm": 0.6540615887428782,
"kl": 0.69287109375,
"learning_rate": 1e-06,
"loss": -0.0011,
"reward": 0.8437500298023224,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.8437500298023224,
"step": 571
},
{
"clip_ratio": 0.0002021094405790791,
"epoch": 95.27586206896552,
"grad_norm": 0.32638937252143785,
"kl": 0.451904296875,
"learning_rate": 1e-06,
"loss": -0.002,
"step": 572
},
{
"batch_accuracy": 0.84375,
"clip_ratio": 0.0,
"completion_length": 290.8259048461914,
"epoch": 95.41379310344827,
"grad_norm": 0.32851949325389684,
"kl": 0.2333984375,
"learning_rate": 1e-06,
"loss": -0.02,
"reward": 0.8437500447034836,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.8437500447034836,
"step": 573
},
{
"clip_ratio": 0.00030660181073471904,
"epoch": 95.55172413793103,
"grad_norm": 0.20845992507237093,
"kl": 0.2421875,
"learning_rate": 1e-06,
"loss": -0.0205,
"step": 574
},
{
"batch_accuracy": 0.9241071428571428,
"clip_ratio": 0.0,
"completion_length": 217.3259048461914,
"epoch": 95.6896551724138,
"grad_norm": 0.20862181127671206,
"kl": 0.259765625,
"learning_rate": 1e-06,
"loss": -0.0016,
"reward": 0.924107164144516,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.924107164144516,
"step": 575
},
{
"clip_ratio": 5.621346281259321e-05,
"epoch": 95.82758620689656,
"grad_norm": 0.15315562157572665,
"kl": 0.22265625,
"learning_rate": 1e-06,
"loss": -0.002,
"step": 576
},
{
"batch_accuracy": 0.9464285714285714,
"clip_ratio": 0.0,
"completion_length": 255.55358505249023,
"epoch": 96.13793103448276,
"grad_norm": 0.6579042856320471,
"kl": 0.35009765625,
"learning_rate": 1e-06,
"loss": -0.005,
"reward": 0.9464286118745804,
"reward_std": 0.04178631864488125,
"rewards/unified_reward_func": 0.9464286118745804,
"step": 577
},
{
"clip_ratio": 0.0008257225781562738,
"epoch": 96.27586206896552,
"grad_norm": 0.35359186008618476,
"kl": 0.25634765625,
"learning_rate": 1e-06,
"loss": -0.0058,
"step": 578
},
{
"batch_accuracy": 0.8080357142857143,
"clip_ratio": 0.0,
"completion_length": 254.8482322692871,
"epoch": 96.41379310344827,
"grad_norm": 0.41082852610725706,
"kl": 0.209716796875,
"learning_rate": 1e-06,
"loss": 0.0097,
"reward": 0.808035746216774,
"reward_std": 0.029159409925341606,
"rewards/unified_reward_func": 0.808035746216774,
"step": 579
},
{
"clip_ratio": 0.00018716741033131257,
"epoch": 96.55172413793103,
"grad_norm": 0.2914419208471443,
"kl": 0.218994140625,
"learning_rate": 1e-06,
"loss": 0.009,
"step": 580
},
{
"batch_accuracy": 0.9910714285714286,
"clip_ratio": 0.0,
"completion_length": 255.79018783569336,
"epoch": 96.6896551724138,
"grad_norm": 0.5785955554778244,
"kl": 0.16015625,
"learning_rate": 1e-06,
"loss": -0.0003,
"reward": 0.9910714626312256,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9910714626312256,
"step": 581
},
{
"clip_ratio": 7.487271068384871e-05,
"epoch": 96.82758620689656,
"grad_norm": 0.673263897970261,
"kl": 0.506591796875,
"learning_rate": 1e-06,
"loss": -0.0006,
"step": 582
},
{
"batch_accuracy": 0.9107142857142857,
"clip_ratio": 0.0,
"completion_length": 241.37054443359375,
"epoch": 97.13793103448276,
"grad_norm": 0.3379560163942377,
"kl": 0.175537109375,
"learning_rate": 1e-06,
"loss": -0.0009,
"reward": 0.9107143431901932,
"reward_std": 0.03111080639064312,
"rewards/unified_reward_func": 0.9107143431901932,
"step": 583
},
{
"clip_ratio": 0.0002795299296849407,
"epoch": 97.27586206896552,
"grad_norm": 0.23988831112022546,
"kl": 0.18505859375,
"learning_rate": 1e-06,
"loss": -0.0016,
"step": 584
},
{
"batch_accuracy": 0.7991071428571428,
"clip_ratio": 0.0,
"completion_length": 274.97769927978516,
"epoch": 97.41379310344827,
"grad_norm": 0.504319613213112,
"kl": 0.41162109375,
"learning_rate": 1e-06,
"loss": -0.0038,
"reward": 0.7991071939468384,
"reward_std": 0.05441322363913059,
"rewards/unified_reward_func": 0.7991071939468384,
"step": 585
},
{
"clip_ratio": 0.0004081932838744251,
"epoch": 97.55172413793103,
"grad_norm": 0.40195943010535184,
"kl": 0.310302734375,
"learning_rate": 1e-06,
"loss": -0.0046,
"step": 586
},
{
"batch_accuracy": 0.8839285714285714,
"clip_ratio": 0.0,
"completion_length": 304.45983505249023,
"epoch": 97.6896551724138,
"grad_norm": 153.18400076986597,
"kl": 27.049072265625,
"learning_rate": 1e-06,
"loss": 0.0295,
"reward": 0.8839285969734192,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.8839285969734192,
"step": 587
},
{
"clip_ratio": 0.00023562640853924677,
"epoch": 97.82758620689656,
"grad_norm": 0.2820135906671078,
"kl": 0.3720703125,
"learning_rate": 1e-06,
"loss": 0.0028,
"step": 588
},
{
"batch_accuracy": 0.9508928571428571,
"clip_ratio": 0.0,
"completion_length": 296.3303756713867,
"epoch": 98.13793103448276,
"grad_norm": 0.3651174815481574,
"kl": 0.28564453125,
"learning_rate": 1e-06,
"loss": -0.0003,
"reward": 0.9508928954601288,
"reward_std": 0.03788072057068348,
"rewards/unified_reward_func": 0.9508928954601288,
"step": 589
},
{
"clip_ratio": 0.00031237147777574137,
"epoch": 98.27586206896552,
"grad_norm": 0.2478077249618373,
"kl": 0.2294921875,
"learning_rate": 1e-06,
"loss": -0.0009,
"step": 590
},
{
"batch_accuracy": 0.8348214285714286,
"clip_ratio": 0.0,
"completion_length": 271.37500762939453,
"epoch": 98.41379310344827,
"grad_norm": 0.5379304215920322,
"kl": 0.42138671875,
"learning_rate": 1e-06,
"loss": -0.0008,
"reward": 0.8348214626312256,
"reward_std": 0.05441322363913059,
"rewards/unified_reward_func": 0.8348214626312256,
"step": 591
},
{
"clip_ratio": 0.0003275511880929116,
"epoch": 98.55172413793103,
"grad_norm": 11.095997641379615,
"kl": 0.21484375,
"learning_rate": 1e-06,
"loss": 0.0037,
"step": 592
},
{
"batch_accuracy": 0.90625,
"clip_ratio": 0.0,
"completion_length": 258.28126525878906,
"epoch": 98.6896551724138,
"grad_norm": 0.5801422636522526,
"kl": 0.224365234375,
"learning_rate": 1e-06,
"loss": 0.0021,
"reward": 0.9062500447034836,
"reward_std": 0.06313453242182732,
"rewards/unified_reward_func": 0.9062500447034836,
"step": 593
},
{
"clip_ratio": 0.0003545371364452876,
"epoch": 98.82758620689656,
"grad_norm": 0.6215554312278989,
"kl": 0.388916015625,
"learning_rate": 1e-06,
"loss": 0.0014,
"step": 594
},
{
"batch_accuracy": 0.8169642857142857,
"clip_ratio": 0.0,
"completion_length": 276.9107246398926,
"epoch": 99.13793103448276,
"grad_norm": 6.009087004919584,
"kl": 1.102294921875,
"learning_rate": 1e-06,
"loss": -0.0008,
"reward": 0.8169643133878708,
"reward_std": 0.012626906856894493,
"rewards/unified_reward_func": 0.8169643133878708,
"step": 595
},
{
"clip_ratio": 9.897070412989706e-05,
"epoch": 99.27586206896552,
"grad_norm": 0.14876236532911624,
"kl": 0.203125,
"learning_rate": 1e-06,
"loss": -0.0017,
"step": 596
},
{
"batch_accuracy": 0.9196428571428571,
"clip_ratio": 0.0,
"completion_length": 260.433048248291,
"epoch": 99.41379310344827,
"grad_norm": 0.25737299250337803,
"kl": 0.283447265625,
"learning_rate": 1e-06,
"loss": -0.0059,
"reward": 0.9196428954601288,
"reward_std": 0.025253813713788986,
"rewards/unified_reward_func": 0.9196428954601288,
"step": 597
},
{
"clip_ratio": 0.00015474966494366527,
"epoch": 99.55172413793103,
"grad_norm": 0.17163873622166492,
"kl": 0.310546875,
"learning_rate": 1e-06,
"loss": -0.0063,
"step": 598
},
{
"batch_accuracy": 0.9107142857142857,
"clip_ratio": 0.0,
"completion_length": 285.45537185668945,
"epoch": 99.6896551724138,
"grad_norm": 0.4237895842134686,
"kl": 0.296630859375,
"learning_rate": 1e-06,
"loss": 0.003,
"reward": 0.9107142984867096,
"reward_std": 0.04178631864488125,
"rewards/unified_reward_func": 0.9107142984867096,
"step": 599
},
{
"epoch": 99.82758620689656,
"grad_norm": 0.33686343212751796,
"learning_rate": 1e-06,
"loss": 0.0023,
"step": 600
},
{
"epoch": 99.82758620689656,
"eval_batch_accuracy": 0.75,
"eval_clip_ratio": 0.0,
"eval_completion_length": 292.19373779296876,
"eval_kl": 0.08203125,
"eval_loss": -0.002871450036764145,
"eval_reward": 0.7500000357627868,
"eval_reward_std": 0.11066367477178574,
"eval_rewards/unified_reward_func": 0.7500000357627868,
"eval_runtime": 317.6626,
"eval_samples_per_second": 0.315,
"eval_steps_per_second": 0.006,
"step": 600
}
],
"logging_steps": 1,
"max_steps": 700,
"num_input_tokens_seen": 0,
"num_train_epochs": 100,
"save_steps": 100,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 0.0,
"train_batch_size": 8,
"trial_name": null,
"trial_params": null
}