| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 99.82758620689656, |
| "eval_steps": 50, |
| "global_step": 600, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "batch_accuracy": 0.4375, |
| "clip_ratio": 0.0, |
| "completion_length": 245.16518783569336, |
| "epoch": 0.13793103448275862, |
| "grad_norm": 9.47688737915962, |
| "kl": 4.5859375, |
| "learning_rate": 0.0, |
| "loss": -0.0177, |
| "reward": 0.4375000260770321, |
| "reward_std": 0.18501290306448936, |
| "rewards/unified_reward_func": 0.4375000260770321, |
| "step": 1 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 0.27586206896551724, |
| "grad_norm": 9.476790609815902, |
| "kl": 4.5859375, |
| "learning_rate": 5e-08, |
| "loss": -0.0177, |
| "step": 2 |
| }, |
| { |
| "batch_accuracy": 0.6473214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 257.0803680419922, |
| "epoch": 0.41379310344827586, |
| "grad_norm": 5.633209761896716, |
| "kl": 3.904296875, |
| "learning_rate": 1e-07, |
| "loss": 0.0136, |
| "reward": 0.6473214626312256, |
| "reward_std": 0.17751124873757362, |
| "rewards/unified_reward_func": 0.6473214626312256, |
| "step": 3 |
| }, |
| { |
| "clip_ratio": 0.004003038338851184, |
| "epoch": 0.5517241379310345, |
| "grad_norm": 5.597864810640234, |
| "kl": 3.912109375, |
| "learning_rate": 1.5e-07, |
| "loss": 0.014, |
| "step": 4 |
| }, |
| { |
| "batch_accuracy": 0.6651785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 262.24108123779297, |
| "epoch": 0.6896551724137931, |
| "grad_norm": 2.1149925603667534, |
| "kl": 1.17431640625, |
| "learning_rate": 2e-07, |
| "loss": -0.0038, |
| "reward": 0.6651786118745804, |
| "reward_std": 0.1720893383026123, |
| "rewards/unified_reward_func": 0.6651786118745804, |
| "step": 5 |
| }, |
| { |
| "clip_ratio": 0.002543629379943013, |
| "epoch": 0.8275862068965517, |
| "grad_norm": 2.0967745522222123, |
| "kl": 1.1083984375, |
| "learning_rate": 2.5e-07, |
| "loss": -0.004, |
| "step": 6 |
| }, |
| { |
| "batch_accuracy": 0.53125, |
| "clip_ratio": 0.0, |
| "completion_length": 253.30805206298828, |
| "epoch": 1.1379310344827587, |
| "grad_norm": 4.155536879447195, |
| "kl": 2.4609375, |
| "learning_rate": 3e-07, |
| "loss": -0.0355, |
| "reward": 0.5312500149011612, |
| "reward_std": 0.18532243371009827, |
| "rewards/unified_reward_func": 0.5312500149011612, |
| "step": 7 |
| }, |
| { |
| "clip_ratio": 0.0036200762260705233, |
| "epoch": 1.2758620689655173, |
| "grad_norm": 4.198764999207007, |
| "kl": 2.4296875, |
| "learning_rate": 3.5e-07, |
| "loss": -0.0353, |
| "step": 8 |
| }, |
| { |
| "batch_accuracy": 0.5669642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 250.44643783569336, |
| "epoch": 1.4137931034482758, |
| "grad_norm": 2.9550024325113653, |
| "kl": 1.6478271484375, |
| "learning_rate": 4e-07, |
| "loss": -0.0255, |
| "reward": 0.5669642984867096, |
| "reward_std": 0.14188865199685097, |
| "rewards/unified_reward_func": 0.5669642984867096, |
| "step": 9 |
| }, |
| { |
| "clip_ratio": 0.0023318197927437723, |
| "epoch": 1.5517241379310345, |
| "grad_norm": 2.9562543058271067, |
| "kl": 1.7484130859375, |
| "learning_rate": 4.5e-07, |
| "loss": -0.0254, |
| "step": 10 |
| }, |
| { |
| "batch_accuracy": 0.5491071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 282.7053756713867, |
| "epoch": 1.6896551724137931, |
| "grad_norm": 396.7503114836838, |
| "kl": 7.328125, |
| "learning_rate": 5e-07, |
| "loss": 0.0191, |
| "reward": 0.5491071715950966, |
| "reward_std": 0.14549032226204872, |
| "rewards/unified_reward_func": 0.5491071715950966, |
| "step": 11 |
| }, |
| { |
| "clip_ratio": 0.0031792108493391424, |
| "epoch": 1.8275862068965516, |
| "grad_norm": 956.7023024380353, |
| "kl": 13.0654296875, |
| "learning_rate": 5.5e-07, |
| "loss": 0.0253, |
| "step": 12 |
| }, |
| { |
| "batch_accuracy": 0.5133928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 308.8884048461914, |
| "epoch": 2.1379310344827585, |
| "grad_norm": 232.27657880882657, |
| "kl": 4.3515625, |
| "learning_rate": 6e-07, |
| "loss": -0.0097, |
| "reward": 0.513392873108387, |
| "reward_std": 0.24723730236291885, |
| "rewards/unified_reward_func": 0.513392873108387, |
| "step": 13 |
| }, |
| { |
| "clip_ratio": 0.0059004299400839955, |
| "epoch": 2.2758620689655173, |
| "grad_norm": 879.2933553212865, |
| "kl": 10.41015625, |
| "learning_rate": 6.5e-07, |
| "loss": -0.0028, |
| "step": 14 |
| }, |
| { |
| "batch_accuracy": 0.5714285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 265.87054443359375, |
| "epoch": 2.413793103448276, |
| "grad_norm": 2255.3480730668302, |
| "kl": 68.51708984375, |
| "learning_rate": 7e-07, |
| "loss": 0.0654, |
| "reward": 0.5714285969734192, |
| "reward_std": 0.17464973405003548, |
| "rewards/unified_reward_func": 0.5714285969734192, |
| "step": 15 |
| }, |
| { |
| "clip_ratio": 0.005311834625899792, |
| "epoch": 2.5517241379310347, |
| "grad_norm": 3452.5379008042123, |
| "kl": 88.62890625, |
| "learning_rate": 7.5e-07, |
| "loss": 0.0857, |
| "step": 16 |
| }, |
| { |
| "batch_accuracy": 0.625, |
| "clip_ratio": 0.0, |
| "completion_length": 211.33036422729492, |
| "epoch": 2.689655172413793, |
| "grad_norm": 1980.5043731662072, |
| "kl": 26.06103515625, |
| "learning_rate": 8e-07, |
| "loss": 0.0318, |
| "reward": 0.6250000223517418, |
| "reward_std": 0.1202336996793747, |
| "rewards/unified_reward_func": 0.6250000223517418, |
| "step": 17 |
| }, |
| { |
| "clip_ratio": 0.003107672091573477, |
| "epoch": 2.8275862068965516, |
| "grad_norm": 319.9609460610652, |
| "kl": 6.7763671875, |
| "learning_rate": 8.499999999999999e-07, |
| "loss": 0.014, |
| "step": 18 |
| }, |
| { |
| "batch_accuracy": 0.5491071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 256.5535888671875, |
| "epoch": 3.1379310344827585, |
| "grad_norm": 14.147026813832152, |
| "kl": 1.04296875, |
| "learning_rate": 9e-07, |
| "loss": -0.007, |
| "reward": 0.549107164144516, |
| "reward_std": 0.20905713737010956, |
| "rewards/unified_reward_func": 0.549107164144516, |
| "step": 19 |
| }, |
| { |
| "clip_ratio": 0.00605745759094134, |
| "epoch": 3.2758620689655173, |
| "grad_norm": 16.315015906851375, |
| "kl": 3.7421875, |
| "learning_rate": 9.499999999999999e-07, |
| "loss": -0.0025, |
| "step": 20 |
| }, |
| { |
| "batch_accuracy": 0.5758928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 254.38394165039062, |
| "epoch": 3.413793103448276, |
| "grad_norm": 401.95809547955486, |
| "kl": 27.5390625, |
| "learning_rate": 1e-06, |
| "loss": 0.0134, |
| "reward": 0.5758928805589676, |
| "reward_std": 0.1720893420279026, |
| "rewards/unified_reward_func": 0.5758928805589676, |
| "step": 21 |
| }, |
| { |
| "clip_ratio": 0.00354966675513424, |
| "epoch": 3.5517241379310347, |
| "grad_norm": 4794.277994865746, |
| "kl": 2.33935546875, |
| "learning_rate": 1e-06, |
| "loss": 0.348, |
| "step": 22 |
| }, |
| { |
| "batch_accuracy": 0.6383928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 232.91072845458984, |
| "epoch": 3.689655172413793, |
| "grad_norm": 9.457388919441831, |
| "kl": 5.744140625, |
| "learning_rate": 1e-06, |
| "loss": -0.0543, |
| "reward": 0.6383928805589676, |
| "reward_std": 0.20905713737010956, |
| "rewards/unified_reward_func": 0.6383928805589676, |
| "step": 23 |
| }, |
| { |
| "clip_ratio": 0.0028303218714427203, |
| "epoch": 3.8275862068965516, |
| "grad_norm": 3.7206165696876052, |
| "kl": 5.05078125, |
| "learning_rate": 1e-06, |
| "loss": -0.0549, |
| "step": 24 |
| }, |
| { |
| "batch_accuracy": 0.5491071428571429, |
| "clip_ratio": 0.0, |
| "completion_length": 266.6696548461914, |
| "epoch": 4.137931034482759, |
| "grad_norm": 3.893488451423076, |
| "kl": 1.9423828125, |
| "learning_rate": 1e-06, |
| "loss": -0.0473, |
| "reward": 0.5491071715950966, |
| "reward_std": 0.27956050261855125, |
| "rewards/unified_reward_func": 0.5491071715950966, |
| "step": 25 |
| }, |
| { |
| "clip_ratio": 0.0044884231756441295, |
| "epoch": 4.275862068965517, |
| "grad_norm": 4.4750395499719, |
| "kl": 1.837890625, |
| "learning_rate": 1e-06, |
| "loss": -0.0474, |
| "step": 26 |
| }, |
| { |
| "batch_accuracy": 0.5982142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 245.59375762939453, |
| "epoch": 4.413793103448276, |
| "grad_norm": 7.430927491828534, |
| "kl": 3.63671875, |
| "learning_rate": 1e-06, |
| "loss": 0.005, |
| "reward": 0.5982142984867096, |
| "reward_std": 0.10114361345767975, |
| "rewards/unified_reward_func": 0.5982142984867096, |
| "step": 27 |
| }, |
| { |
| "clip_ratio": 0.0019684870640048757, |
| "epoch": 4.551724137931035, |
| "grad_norm": 6.857779970270849, |
| "kl": 3.32568359375, |
| "learning_rate": 1e-06, |
| "loss": 0.0049, |
| "step": 28 |
| }, |
| { |
| "batch_accuracy": 0.5446428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 252.18750762939453, |
| "epoch": 4.689655172413794, |
| "grad_norm": 5.538508182062988, |
| "kl": 2.44140625, |
| "learning_rate": 1e-06, |
| "loss": -0.028, |
| "reward": 0.5446428805589676, |
| "reward_std": 0.12895501032471657, |
| "rewards/unified_reward_func": 0.5446428805589676, |
| "step": 29 |
| }, |
| { |
| "clip_ratio": 0.002361440951062832, |
| "epoch": 4.827586206896552, |
| "grad_norm": 28.554258518690066, |
| "kl": 1.544921875, |
| "learning_rate": 1e-06, |
| "loss": -0.0252, |
| "step": 30 |
| }, |
| { |
| "batch_accuracy": 0.6116071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 255.02679824829102, |
| "epoch": 5.137931034482759, |
| "grad_norm": 27.41270780548312, |
| "kl": 1.9775390625, |
| "learning_rate": 1e-06, |
| "loss": 0.0004, |
| "reward": 0.611607164144516, |
| "reward_std": 0.1989906169474125, |
| "rewards/unified_reward_func": 0.611607164144516, |
| "step": 31 |
| }, |
| { |
| "clip_ratio": 0.005010936001781374, |
| "epoch": 5.275862068965517, |
| "grad_norm": 351.5618938932037, |
| "kl": 1.869140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0131, |
| "step": 32 |
| }, |
| { |
| "batch_accuracy": 0.5267857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 255.38840103149414, |
| "epoch": 5.413793103448276, |
| "grad_norm": 221.7307789828897, |
| "kl": 4.8828125, |
| "learning_rate": 1e-06, |
| "loss": -0.0152, |
| "reward": 0.5267857387661934, |
| "reward_std": 0.15872061625123024, |
| "rewards/unified_reward_func": 0.5267857387661934, |
| "step": 33 |
| }, |
| { |
| "clip_ratio": 0.003642797644715756, |
| "epoch": 5.551724137931035, |
| "grad_norm": 19.906197185838778, |
| "kl": 1.984375, |
| "learning_rate": 1e-06, |
| "loss": -0.0163, |
| "step": 34 |
| }, |
| { |
| "batch_accuracy": 0.71875, |
| "clip_ratio": 0.0, |
| "completion_length": 231.86608505249023, |
| "epoch": 5.689655172413794, |
| "grad_norm": 425.6065826516748, |
| "kl": 13.2099609375, |
| "learning_rate": 1e-06, |
| "loss": -0.0066, |
| "reward": 0.7187500447034836, |
| "reward_std": 0.15676921978592873, |
| "rewards/unified_reward_func": 0.7187500447034836, |
| "step": 35 |
| }, |
| { |
| "clip_ratio": 0.004165306163486093, |
| "epoch": 5.827586206896552, |
| "grad_norm": 1283.2157555121717, |
| "kl": 25.34765625, |
| "learning_rate": 1e-06, |
| "loss": 0.0071, |
| "step": 36 |
| }, |
| { |
| "batch_accuracy": 0.5625, |
| "clip_ratio": 0.0, |
| "completion_length": 243.47322463989258, |
| "epoch": 6.137931034482759, |
| "grad_norm": 476.5962581699318, |
| "kl": 12.490234375, |
| "learning_rate": 1e-06, |
| "loss": -0.0032, |
| "reward": 0.5625000298023224, |
| "reward_std": 0.16531942039728165, |
| "rewards/unified_reward_func": 0.5625000298023224, |
| "step": 37 |
| }, |
| { |
| "clip_ratio": 0.003403076552785933, |
| "epoch": 6.275862068965517, |
| "grad_norm": 58.50488278217316, |
| "kl": 4.296875, |
| "learning_rate": 1e-06, |
| "loss": -0.0074, |
| "step": 38 |
| }, |
| { |
| "batch_accuracy": 0.5803571428571429, |
| "clip_ratio": 0.0, |
| "completion_length": 279.6562690734863, |
| "epoch": 6.413793103448276, |
| "grad_norm": 20.57670369517026, |
| "kl": 1.1103515625, |
| "learning_rate": 1e-06, |
| "loss": 0.0094, |
| "reward": 0.580357164144516, |
| "reward_std": 0.18245531991124153, |
| "rewards/unified_reward_func": 0.580357164144516, |
| "step": 39 |
| }, |
| { |
| "clip_ratio": 0.003295150410849601, |
| "epoch": 6.551724137931035, |
| "grad_norm": 123.77326026058726, |
| "kl": 0.611328125, |
| "learning_rate": 1e-06, |
| "loss": 0.013, |
| "step": 40 |
| }, |
| { |
| "batch_accuracy": 0.6741071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 284.7366142272949, |
| "epoch": 6.689655172413794, |
| "grad_norm": 27.593321692714238, |
| "kl": 1.416015625, |
| "learning_rate": 1e-06, |
| "loss": -0.0192, |
| "reward": 0.6741071790456772, |
| "reward_std": 0.14413951337337494, |
| "rewards/unified_reward_func": 0.6741071790456772, |
| "step": 41 |
| }, |
| { |
| "clip_ratio": 0.0034483993076719344, |
| "epoch": 6.827586206896552, |
| "grad_norm": 9.09534191384267, |
| "kl": 1.541015625, |
| "learning_rate": 1e-06, |
| "loss": -0.0186, |
| "step": 42 |
| }, |
| { |
| "batch_accuracy": 0.6160714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 271.86609268188477, |
| "epoch": 7.137931034482759, |
| "grad_norm": 4.7592529588927315, |
| "kl": 1.0712890625, |
| "learning_rate": 1e-06, |
| "loss": -0.0101, |
| "reward": 0.6160714477300644, |
| "reward_std": 0.20185214653611183, |
| "rewards/unified_reward_func": 0.6160714477300644, |
| "step": 43 |
| }, |
| { |
| "clip_ratio": 0.003981543180998415, |
| "epoch": 7.275862068965517, |
| "grad_norm": 16.951873635923167, |
| "kl": 0.806640625, |
| "learning_rate": 1e-06, |
| "loss": -0.0106, |
| "step": 44 |
| }, |
| { |
| "batch_accuracy": 0.6383928571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 228.32144165039062, |
| "epoch": 7.413793103448276, |
| "grad_norm": 8.354185239868439, |
| "kl": 0.94140625, |
| "learning_rate": 1e-06, |
| "loss": -0.02, |
| "reward": 0.6383928954601288, |
| "reward_std": 0.21417231857776642, |
| "rewards/unified_reward_func": 0.6383928954601288, |
| "step": 45 |
| }, |
| { |
| "clip_ratio": 0.002367985143791884, |
| "epoch": 7.551724137931035, |
| "grad_norm": 8.373422170690585, |
| "kl": 0.888671875, |
| "learning_rate": 1e-06, |
| "loss": -0.0204, |
| "step": 46 |
| }, |
| { |
| "batch_accuracy": 0.5669642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 287.12500762939453, |
| "epoch": 7.689655172413794, |
| "grad_norm": 21.451464508489934, |
| "kl": 0.85693359375, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "reward": 0.5669643133878708, |
| "reward_std": 0.1665289979428053, |
| "rewards/unified_reward_func": 0.5669643133878708, |
| "step": 47 |
| }, |
| { |
| "clip_ratio": 0.002784289186820388, |
| "epoch": 7.827586206896552, |
| "grad_norm": 32.16378042731212, |
| "kl": 1.087890625, |
| "learning_rate": 1e-06, |
| "loss": 0.0011, |
| "step": 48 |
| }, |
| { |
| "batch_accuracy": 0.6696428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 251.9107208251953, |
| "epoch": 8.137931034482758, |
| "grad_norm": 15.579813273195894, |
| "kl": 0.83251953125, |
| "learning_rate": 1e-06, |
| "loss": -0.0025, |
| "reward": 0.6696428805589676, |
| "reward_std": 0.15464391559362411, |
| "rewards/unified_reward_func": 0.6696428805589676, |
| "step": 49 |
| }, |
| { |
| "epoch": 8.275862068965518, |
| "grad_norm": 1065.9074173474285, |
| "learning_rate": 1e-06, |
| "loss": 0.0116, |
| "step": 50 |
| }, |
| { |
| "epoch": 8.275862068965518, |
| "eval_batch_accuracy": 0.6047619047619047, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 270.27631123860675, |
| "eval_kl": 2.047005208333333, |
| "eval_loss": -0.0024145618081092834, |
| "eval_reward": 0.6047619342803955, |
| "eval_reward_std": 0.1999184677998225, |
| "eval_rewards/unified_reward_func": 0.6047619342803955, |
| "eval_runtime": 336.3398, |
| "eval_samples_per_second": 0.297, |
| "eval_steps_per_second": 0.006, |
| "step": 50 |
| }, |
| { |
| "batch_accuracy": 0.5491071428571428, |
| "clip_ratio": 0.001715210877591744, |
| "completion_length": 287.3884048461914, |
| "epoch": 8.413793103448276, |
| "grad_norm": 42.612573532085165, |
| "kl": 7.935546875, |
| "learning_rate": 1e-06, |
| "loss": -0.003, |
| "reward": 0.5491071566939354, |
| "reward_std": 0.15555683337152004, |
| "rewards/unified_reward_func": 0.5491071566939354, |
| "step": 51 |
| }, |
| { |
| "clip_ratio": 0.002908625523559749, |
| "epoch": 8.551724137931034, |
| "grad_norm": 89.91155792984003, |
| "kl": 0.63134765625, |
| "learning_rate": 1e-06, |
| "loss": -0.0032, |
| "step": 52 |
| }, |
| { |
| "batch_accuracy": 0.6473214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 255.51787185668945, |
| "epoch": 8.689655172413794, |
| "grad_norm": 1.3501585457962537, |
| "kl": 0.81201171875, |
| "learning_rate": 1e-06, |
| "loss": -0.0016, |
| "reward": 0.6473214775323868, |
| "reward_std": 0.16006861999630928, |
| "rewards/unified_reward_func": 0.6473214775323868, |
| "step": 53 |
| }, |
| { |
| "clip_ratio": 0.0019645251450128853, |
| "epoch": 8.827586206896552, |
| "grad_norm": 5.7495234417840075, |
| "kl": 0.70654296875, |
| "learning_rate": 1e-06, |
| "loss": -0.0028, |
| "step": 54 |
| }, |
| { |
| "batch_accuracy": 0.6160714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 285.558048248291, |
| "epoch": 9.137931034482758, |
| "grad_norm": 22.095081988775654, |
| "kl": 0.93408203125, |
| "learning_rate": 1e-06, |
| "loss": 0.0001, |
| "reward": 0.6160714477300644, |
| "reward_std": 0.16006582230329514, |
| "rewards/unified_reward_func": 0.6160714477300644, |
| "step": 55 |
| }, |
| { |
| "clip_ratio": 0.0038310182571876794, |
| "epoch": 9.275862068965518, |
| "grad_norm": 70.73324902480591, |
| "kl": 7.00390625, |
| "learning_rate": 1e-06, |
| "loss": 0.0073, |
| "step": 56 |
| }, |
| { |
| "batch_accuracy": 0.6473214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 251.6428680419922, |
| "epoch": 9.413793103448276, |
| "grad_norm": 10020.221677903297, |
| "kl": 114.0625, |
| "learning_rate": 1e-06, |
| "loss": 0.1067, |
| "reward": 0.6473214626312256, |
| "reward_std": 0.22935960441827774, |
| "rewards/unified_reward_func": 0.6473214626312256, |
| "step": 57 |
| }, |
| { |
| "clip_ratio": 0.005764591624028981, |
| "epoch": 9.551724137931034, |
| "grad_norm": 95.41037910173195, |
| "kl": 5.66015625, |
| "learning_rate": 1e-06, |
| "loss": 0.0077, |
| "step": 58 |
| }, |
| { |
| "batch_accuracy": 0.6651785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 242.2500114440918, |
| "epoch": 9.689655172413794, |
| "grad_norm": 264.24354788959295, |
| "kl": 16.71875, |
| "learning_rate": 1e-06, |
| "loss": 0.0046, |
| "reward": 0.6651786118745804, |
| "reward_std": 0.08552403189241886, |
| "rewards/unified_reward_func": 0.6651786118745804, |
| "step": 59 |
| }, |
| { |
| "clip_ratio": 0.0016965966206043959, |
| "epoch": 9.827586206896552, |
| "grad_norm": 41.8877627907783, |
| "kl": 17.09375, |
| "learning_rate": 1e-06, |
| "loss": 0.005, |
| "step": 60 |
| }, |
| { |
| "batch_accuracy": 0.5580357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 287.27233505249023, |
| "epoch": 10.137931034482758, |
| "grad_norm": 9.649367519026018, |
| "kl": 4.73828125, |
| "learning_rate": 1e-06, |
| "loss": 0.0293, |
| "reward": 0.5580357313156128, |
| "reward_std": 0.18202302604913712, |
| "rewards/unified_reward_func": 0.5580357313156128, |
| "step": 61 |
| }, |
| { |
| "clip_ratio": 0.003987273928942159, |
| "epoch": 10.275862068965518, |
| "grad_norm": 4.26991643094351, |
| "kl": 1.8408203125, |
| "learning_rate": 1e-06, |
| "loss": 0.0287, |
| "step": 62 |
| }, |
| { |
| "batch_accuracy": 0.6875, |
| "clip_ratio": 0.0, |
| "completion_length": 247.79465103149414, |
| "epoch": 10.413793103448276, |
| "grad_norm": 5.240743579266302, |
| "kl": 2.392578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0199, |
| "reward": 0.6875000149011612, |
| "reward_std": 0.165928415954113, |
| "rewards/unified_reward_func": 0.6875000149011612, |
| "step": 63 |
| }, |
| { |
| "clip_ratio": 0.002536454499932006, |
| "epoch": 10.551724137931034, |
| "grad_norm": 0.6982187039233699, |
| "kl": 1.521484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0191, |
| "step": 64 |
| }, |
| { |
| "batch_accuracy": 0.6875, |
| "clip_ratio": 0.0, |
| "completion_length": 264.5446548461914, |
| "epoch": 10.689655172413794, |
| "grad_norm": 1.4881013556571023, |
| "kl": 1.2353515625, |
| "learning_rate": 1e-06, |
| "loss": -0.0058, |
| "reward": 0.6875000298023224, |
| "reward_std": 0.1382853128015995, |
| "rewards/unified_reward_func": 0.6875000298023224, |
| "step": 65 |
| }, |
| { |
| "clip_ratio": 0.003119972941931337, |
| "epoch": 10.827586206896552, |
| "grad_norm": 3.092593425215386, |
| "kl": 0.789794921875, |
| "learning_rate": 1e-06, |
| "loss": -0.0066, |
| "step": 66 |
| }, |
| { |
| "batch_accuracy": 0.6160714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 311.433048248291, |
| "epoch": 11.137931034482758, |
| "grad_norm": 1.01948057937195, |
| "kl": 1.1279296875, |
| "learning_rate": 1e-06, |
| "loss": 0.0049, |
| "reward": 0.6160714477300644, |
| "reward_std": 0.14548752084374428, |
| "rewards/unified_reward_func": 0.6160714477300644, |
| "step": 67 |
| }, |
| { |
| "clip_ratio": 0.002209829064668156, |
| "epoch": 11.275862068965518, |
| "grad_norm": 0.8281537495597667, |
| "kl": 1.1767578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0042, |
| "step": 68 |
| }, |
| { |
| "batch_accuracy": 0.7053571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 250.76340103149414, |
| "epoch": 11.413793103448276, |
| "grad_norm": 0.9634703459739232, |
| "kl": 0.62939453125, |
| "learning_rate": 1e-06, |
| "loss": 0.0402, |
| "reward": 0.705357164144516, |
| "reward_std": 0.16683853790163994, |
| "rewards/unified_reward_func": 0.705357164144516, |
| "step": 69 |
| }, |
| { |
| "clip_ratio": 0.0029377865139395, |
| "epoch": 11.551724137931034, |
| "grad_norm": 1.5188872952272299, |
| "kl": 0.5576171875, |
| "learning_rate": 1e-06, |
| "loss": 0.0387, |
| "step": 70 |
| }, |
| { |
| "batch_accuracy": 0.65625, |
| "clip_ratio": 0.0, |
| "completion_length": 268.31697845458984, |
| "epoch": 11.689655172413794, |
| "grad_norm": 0.9021269984465355, |
| "kl": 0.8798828125, |
| "learning_rate": 1e-06, |
| "loss": -0.0067, |
| "reward": 0.6562500298023224, |
| "reward_std": 0.14097853749990463, |
| "rewards/unified_reward_func": 0.6562500298023224, |
| "step": 71 |
| }, |
| { |
| "clip_ratio": 0.0036392371403053403, |
| "epoch": 11.827586206896552, |
| "grad_norm": 0.9695227093029657, |
| "kl": 1.078125, |
| "learning_rate": 1e-06, |
| "loss": -0.0074, |
| "step": 72 |
| }, |
| { |
| "batch_accuracy": 0.5892857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 246.7187614440918, |
| "epoch": 12.137931034482758, |
| "grad_norm": 7.117785992825863, |
| "kl": 3.7001953125, |
| "learning_rate": 1e-06, |
| "loss": -0.0122, |
| "reward": 0.589285746216774, |
| "reward_std": 0.22229304164648056, |
| "rewards/unified_reward_func": 0.589285746216774, |
| "step": 73 |
| }, |
| { |
| "clip_ratio": 0.006052218785043806, |
| "epoch": 12.275862068965518, |
| "grad_norm": 58.99563589475248, |
| "kl": 1.0908203125, |
| "learning_rate": 1e-06, |
| "loss": -0.0004, |
| "step": 74 |
| }, |
| { |
| "batch_accuracy": 0.7723214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 265.3973274230957, |
| "epoch": 12.413793103448276, |
| "grad_norm": 1.7066461907389874, |
| "kl": 0.826904296875, |
| "learning_rate": 1e-06, |
| "loss": -0.004, |
| "reward": 0.7723214626312256, |
| "reward_std": 0.1415819153189659, |
| "rewards/unified_reward_func": 0.7723214626312256, |
| "step": 75 |
| }, |
| { |
| "clip_ratio": 0.002809168363455683, |
| "epoch": 12.551724137931034, |
| "grad_norm": 1.0796312368974332, |
| "kl": 0.443359375, |
| "learning_rate": 1e-06, |
| "loss": -0.0046, |
| "step": 76 |
| }, |
| { |
| "batch_accuracy": 0.6294642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 276.7232246398926, |
| "epoch": 12.689655172413794, |
| "grad_norm": 1.1083371008590541, |
| "kl": 0.900390625, |
| "learning_rate": 1e-06, |
| "loss": -0.0025, |
| "reward": 0.6294643133878708, |
| "reward_std": 0.14353612437844276, |
| "rewards/unified_reward_func": 0.6294643133878708, |
| "step": 77 |
| }, |
| { |
| "clip_ratio": 0.0021633205469697714, |
| "epoch": 12.827586206896552, |
| "grad_norm": 1.104357377092878, |
| "kl": 0.7685546875, |
| "learning_rate": 1e-06, |
| "loss": -0.0032, |
| "step": 78 |
| }, |
| { |
| "batch_accuracy": 0.7857142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 230.9866180419922, |
| "epoch": 13.137931034482758, |
| "grad_norm": 5.1546027891456845, |
| "kl": 0.912353515625, |
| "learning_rate": 1e-06, |
| "loss": 0.02, |
| "reward": 0.785714328289032, |
| "reward_std": 0.08942962624132633, |
| "rewards/unified_reward_func": 0.785714328289032, |
| "step": 79 |
| }, |
| { |
| "clip_ratio": 0.0016282629658235237, |
| "epoch": 13.275862068965518, |
| "grad_norm": 7.153630585121116, |
| "kl": 0.74951171875, |
| "learning_rate": 1e-06, |
| "loss": 0.0197, |
| "step": 80 |
| }, |
| { |
| "batch_accuracy": 0.4553571428571429, |
| "clip_ratio": 0.0, |
| "completion_length": 332.6651916503906, |
| "epoch": 13.413793103448276, |
| "grad_norm": 3.7470240632805467, |
| "kl": 2.3740234375, |
| "learning_rate": 1e-06, |
| "loss": -0.0099, |
| "reward": 0.4553571678698063, |
| "reward_std": 0.2024611309170723, |
| "rewards/unified_reward_func": 0.4553571678698063, |
| "step": 81 |
| }, |
| { |
| "clip_ratio": 0.003834200673736632, |
| "epoch": 13.551724137931034, |
| "grad_norm": 1.0885699596296923, |
| "kl": 1.330078125, |
| "learning_rate": 1e-06, |
| "loss": -0.0112, |
| "step": 82 |
| }, |
| { |
| "batch_accuracy": 0.7767857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 296.401798248291, |
| "epoch": 13.689655172413794, |
| "grad_norm": 1.094797967613683, |
| "kl": 0.8017578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0183, |
| "reward": 0.776785746216774, |
| "reward_std": 0.1814168505370617, |
| "rewards/unified_reward_func": 0.776785746216774, |
| "step": 83 |
| }, |
| { |
| "clip_ratio": 0.0045213785488158464, |
| "epoch": 13.827586206896552, |
| "grad_norm": 0.8596258608296502, |
| "kl": 0.736328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0173, |
| "step": 84 |
| }, |
| { |
| "batch_accuracy": 0.6696428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 266.92858123779297, |
| "epoch": 14.137931034482758, |
| "grad_norm": 0.6150343367414953, |
| "kl": 0.7158203125, |
| "learning_rate": 1e-06, |
| "loss": 0.0125, |
| "reward": 0.6696428805589676, |
| "reward_std": 0.07875411957502365, |
| "rewards/unified_reward_func": 0.6696428805589676, |
| "step": 85 |
| }, |
| { |
| "clip_ratio": 0.0018555940187070519, |
| "epoch": 14.275862068965518, |
| "grad_norm": 0.5097068454390808, |
| "kl": 0.6337890625, |
| "learning_rate": 1e-06, |
| "loss": 0.0117, |
| "step": 86 |
| }, |
| { |
| "batch_accuracy": 0.6919642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 254.7812614440918, |
| "epoch": 14.413793103448276, |
| "grad_norm": 7.244582112989614, |
| "kl": 3.82568359375, |
| "learning_rate": 1e-06, |
| "loss": 0.0225, |
| "reward": 0.6919643133878708, |
| "reward_std": 0.16336803138256073, |
| "rewards/unified_reward_func": 0.6919643133878708, |
| "step": 87 |
| }, |
| { |
| "clip_ratio": 0.0048985659959726036, |
| "epoch": 14.551724137931034, |
| "grad_norm": 18.06983316871383, |
| "kl": 1.03125, |
| "learning_rate": 1e-06, |
| "loss": 0.0259, |
| "step": 88 |
| }, |
| { |
| "batch_accuracy": 0.6964285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 294.2321586608887, |
| "epoch": 14.689655172413794, |
| "grad_norm": 1.6611859912376654, |
| "kl": 2.0693359375, |
| "learning_rate": 1e-06, |
| "loss": 0.0579, |
| "reward": 0.6964285969734192, |
| "reward_std": 0.16037255339324474, |
| "rewards/unified_reward_func": 0.6964285969734192, |
| "step": 89 |
| }, |
| { |
| "clip_ratio": 0.00386410066857934, |
| "epoch": 14.827586206896552, |
| "grad_norm": 1.1461095674554305, |
| "kl": 1.39453125, |
| "learning_rate": 1e-06, |
| "loss": 0.0566, |
| "step": 90 |
| }, |
| { |
| "batch_accuracy": 0.7544642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 263.4509048461914, |
| "epoch": 15.137931034482758, |
| "grad_norm": 2.7879946254038073, |
| "kl": 4.13623046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0507, |
| "reward": 0.754464328289032, |
| "reward_std": 0.13512154668569565, |
| "rewards/unified_reward_func": 0.754464328289032, |
| "step": 91 |
| }, |
| { |
| "clip_ratio": 0.002846388262696564, |
| "epoch": 15.275862068965518, |
| "grad_norm": 0.8191932494227651, |
| "kl": 1.25732421875, |
| "learning_rate": 1e-06, |
| "loss": 0.0476, |
| "step": 92 |
| }, |
| { |
| "batch_accuracy": 0.7142857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 310.2143020629883, |
| "epoch": 15.413793103448276, |
| "grad_norm": 1.1634079207613266, |
| "kl": 2.3837890625, |
| "learning_rate": 1e-06, |
| "loss": 0.0156, |
| "reward": 0.714285746216774, |
| "reward_std": 0.16006582602858543, |
| "rewards/unified_reward_func": 0.714285746216774, |
| "step": 93 |
| }, |
| { |
| "clip_ratio": 0.0025695697695482522, |
| "epoch": 15.551724137931034, |
| "grad_norm": 0.9306916071768522, |
| "kl": 1.99169921875, |
| "learning_rate": 1e-06, |
| "loss": 0.0145, |
| "step": 94 |
| }, |
| { |
| "batch_accuracy": 0.6830357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 261.5401916503906, |
| "epoch": 15.689655172413794, |
| "grad_norm": 1.2265984060410038, |
| "kl": 1.3271484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0099, |
| "reward": 0.6830357313156128, |
| "reward_std": 0.08222462423145771, |
| "rewards/unified_reward_func": 0.6830357313156128, |
| "step": 95 |
| }, |
| { |
| "clip_ratio": 0.0021528883662540466, |
| "epoch": 15.827586206896552, |
| "grad_norm": 0.7608892170172238, |
| "kl": 0.8759765625, |
| "learning_rate": 1e-06, |
| "loss": 0.009, |
| "step": 96 |
| }, |
| { |
| "batch_accuracy": 0.7276785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 297.6026916503906, |
| "epoch": 16.137931034482758, |
| "grad_norm": 0.9700441408074749, |
| "kl": 1.2021484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0472, |
| "reward": 0.7276786118745804, |
| "reward_std": 0.21192146465182304, |
| "rewards/unified_reward_func": 0.7276786118745804, |
| "step": 97 |
| }, |
| { |
| "clip_ratio": 0.005118858069181442, |
| "epoch": 16.275862068965516, |
| "grad_norm": 1.6631282888053047, |
| "kl": 1.23828125, |
| "learning_rate": 1e-06, |
| "loss": 0.0458, |
| "step": 98 |
| }, |
| { |
| "batch_accuracy": 0.75, |
| "clip_ratio": 0.0, |
| "completion_length": 274.2142906188965, |
| "epoch": 16.413793103448278, |
| "grad_norm": 0.4317146888702674, |
| "kl": 1.02587890625, |
| "learning_rate": 1e-06, |
| "loss": 0.0041, |
| "reward": 0.7500000298023224, |
| "reward_std": 0.09046810120344162, |
| "rewards/unified_reward_func": 0.7500000298023224, |
| "step": 99 |
| }, |
| { |
| "epoch": 16.551724137931036, |
| "grad_norm": 1.133590964829475, |
| "learning_rate": 1e-06, |
| "loss": 0.0034, |
| "step": 100 |
| }, |
| { |
| "epoch": 16.551724137931036, |
| "eval_batch_accuracy": 0.6797619047619048, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 292.35174153645835, |
| "eval_kl": 0.558984375, |
| "eval_loss": -0.0012742819963023067, |
| "eval_reward": 0.6797619382540385, |
| "eval_reward_std": 0.1622819994886716, |
| "eval_rewards/unified_reward_func": 0.6797619382540385, |
| "eval_runtime": 357.5239, |
| "eval_samples_per_second": 0.28, |
| "eval_steps_per_second": 0.006, |
| "step": 100 |
| }, |
| { |
| "batch_accuracy": 0.6428571428571429, |
| "clip_ratio": 0.000845697577460669, |
| "completion_length": 257.42858123779297, |
| "epoch": 16.689655172413794, |
| "grad_norm": 1.2995812903101538, |
| "kl": 0.662841796875, |
| "learning_rate": 1e-06, |
| "loss": 0.018, |
| "reward": 0.6428571790456772, |
| "reward_std": 0.12054043263196945, |
| "rewards/unified_reward_func": 0.6428571790456772, |
| "step": 101 |
| }, |
| { |
| "clip_ratio": 0.0017673625843599439, |
| "epoch": 16.82758620689655, |
| "grad_norm": 0.6537204602789219, |
| "kl": 0.349853515625, |
| "learning_rate": 1e-06, |
| "loss": 0.0174, |
| "step": 102 |
| }, |
| { |
| "batch_accuracy": 0.7276785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 296.745548248291, |
| "epoch": 17.137931034482758, |
| "grad_norm": 0.9855908928682507, |
| "kl": 0.4176025390625, |
| "learning_rate": 1e-06, |
| "loss": 0.0449, |
| "reward": 0.7276785969734192, |
| "reward_std": 0.15555684082210064, |
| "rewards/unified_reward_func": 0.7276785969734192, |
| "step": 103 |
| }, |
| { |
| "clip_ratio": 0.004551950143650174, |
| "epoch": 17.275862068965516, |
| "grad_norm": 5.015002215768227, |
| "kl": 0.46875, |
| "learning_rate": 1e-06, |
| "loss": 0.0437, |
| "step": 104 |
| }, |
| { |
| "batch_accuracy": 0.7455357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 269.9464416503906, |
| "epoch": 17.413793103448278, |
| "grad_norm": 0.5358380953757037, |
| "kl": 0.333984375, |
| "learning_rate": 1e-06, |
| "loss": 0.0016, |
| "reward": 0.7455357611179352, |
| "reward_std": 0.08265971392393112, |
| "rewards/unified_reward_func": 0.7455357611179352, |
| "step": 105 |
| }, |
| { |
| "clip_ratio": 0.001166727059171535, |
| "epoch": 17.551724137931036, |
| "grad_norm": 0.6830961806380172, |
| "kl": 0.297119140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0008, |
| "step": 106 |
| }, |
| { |
| "batch_accuracy": 0.7142857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 306.00001525878906, |
| "epoch": 17.689655172413794, |
| "grad_norm": 0.6823238907681224, |
| "kl": 0.356689453125, |
| "learning_rate": 1e-06, |
| "loss": 0.0279, |
| "reward": 0.7142857313156128, |
| "reward_std": 0.1010152529925108, |
| "rewards/unified_reward_func": 0.7142857313156128, |
| "step": 107 |
| }, |
| { |
| "clip_ratio": 0.002543082577176392, |
| "epoch": 17.82758620689655, |
| "grad_norm": 0.6313266960191072, |
| "kl": 0.359619140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0269, |
| "step": 108 |
| }, |
| { |
| "batch_accuracy": 0.7321428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 287.0044746398926, |
| "epoch": 18.137931034482758, |
| "grad_norm": 0.8019490958341388, |
| "kl": 0.649169921875, |
| "learning_rate": 1e-06, |
| "loss": 0.0058, |
| "reward": 0.7321428805589676, |
| "reward_std": 0.13902713917195797, |
| "rewards/unified_reward_func": 0.7321428805589676, |
| "step": 109 |
| }, |
| { |
| "clip_ratio": 0.002602686858153902, |
| "epoch": 18.275862068965516, |
| "grad_norm": 0.6107705837722435, |
| "kl": 0.675537109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0047, |
| "step": 110 |
| }, |
| { |
| "batch_accuracy": 0.7678571428571429, |
| "clip_ratio": 0.0, |
| "completion_length": 279.4821548461914, |
| "epoch": 18.413793103448278, |
| "grad_norm": 0.6770932627526579, |
| "kl": 0.353515625, |
| "learning_rate": 1e-06, |
| "loss": -0.0067, |
| "reward": 0.7678571790456772, |
| "reward_std": 0.10851971805095673, |
| "rewards/unified_reward_func": 0.7678571790456772, |
| "step": 111 |
| }, |
| { |
| "clip_ratio": 0.0023510658502345905, |
| "epoch": 18.551724137931036, |
| "grad_norm": 0.7583576047177788, |
| "kl": 0.293701171875, |
| "learning_rate": 1e-06, |
| "loss": -0.0072, |
| "step": 112 |
| }, |
| { |
| "batch_accuracy": 0.75, |
| "clip_ratio": 0.0, |
| "completion_length": 287.3214416503906, |
| "epoch": 18.689655172413794, |
| "grad_norm": 0.5279948952585755, |
| "kl": 0.51806640625, |
| "learning_rate": 1e-06, |
| "loss": 0.0003, |
| "reward": 0.7500000447034836, |
| "reward_std": 0.07875411584973335, |
| "rewards/unified_reward_func": 0.7500000447034836, |
| "step": 113 |
| }, |
| { |
| "clip_ratio": 0.0013883734354749322, |
| "epoch": 18.82758620689655, |
| "grad_norm": 0.38877289023644157, |
| "kl": 0.470703125, |
| "learning_rate": 1e-06, |
| "loss": -0.0002, |
| "step": 114 |
| }, |
| { |
| "batch_accuracy": 0.71875, |
| "clip_ratio": 0.0, |
| "completion_length": 271.07590103149414, |
| "epoch": 19.137931034482758, |
| "grad_norm": 0.7346182063865974, |
| "kl": 0.3056640625, |
| "learning_rate": 1e-06, |
| "loss": 0.0083, |
| "reward": 0.7187500447034836, |
| "reward_std": 0.12956120818853378, |
| "rewards/unified_reward_func": 0.7187500447034836, |
| "step": 115 |
| }, |
| { |
| "clip_ratio": 0.003321638738270849, |
| "epoch": 19.275862068965516, |
| "grad_norm": 0.7155226380704629, |
| "kl": 0.312255859375, |
| "learning_rate": 1e-06, |
| "loss": 0.0074, |
| "step": 116 |
| }, |
| { |
| "batch_accuracy": 0.7901785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 282.6250114440918, |
| "epoch": 19.413793103448278, |
| "grad_norm": 0.764157190415015, |
| "kl": 0.69677734375, |
| "learning_rate": 1e-06, |
| "loss": 0.0284, |
| "reward": 0.7901785969734192, |
| "reward_std": 0.07350331731140614, |
| "rewards/unified_reward_func": 0.7901785969734192, |
| "step": 117 |
| }, |
| { |
| "clip_ratio": 0.002791708306176588, |
| "epoch": 19.551724137931036, |
| "grad_norm": 0.9985727394911575, |
| "kl": 0.6328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0277, |
| "step": 118 |
| }, |
| { |
| "batch_accuracy": 0.8392857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 250.7812614440918, |
| "epoch": 19.689655172413794, |
| "grad_norm": 0.6371173760577464, |
| "kl": 0.450439453125, |
| "learning_rate": 1e-06, |
| "loss": -0.001, |
| "reward": 0.839285746216774, |
| "reward_std": 0.0835726372897625, |
| "rewards/unified_reward_func": 0.839285746216774, |
| "step": 119 |
| }, |
| { |
| "clip_ratio": 0.0017081814585253596, |
| "epoch": 19.82758620689655, |
| "grad_norm": 0.49119996087886486, |
| "kl": 0.41259765625, |
| "learning_rate": 1e-06, |
| "loss": -0.0016, |
| "step": 120 |
| }, |
| { |
| "batch_accuracy": 0.8392857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 265.5759048461914, |
| "epoch": 20.137931034482758, |
| "grad_norm": 0.8976181584128184, |
| "kl": 0.83349609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0186, |
| "reward": 0.839285746216774, |
| "reward_std": 0.0754547119140625, |
| "rewards/unified_reward_func": 0.839285746216774, |
| "step": 121 |
| }, |
| { |
| "clip_ratio": 0.0017390022403560579, |
| "epoch": 20.275862068965516, |
| "grad_norm": 0.5861725820573414, |
| "kl": 0.5615234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0178, |
| "step": 122 |
| }, |
| { |
| "batch_accuracy": 0.7857142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 263.714298248291, |
| "epoch": 20.413793103448278, |
| "grad_norm": 0.46434375354901414, |
| "kl": 0.4921875, |
| "learning_rate": 1e-06, |
| "loss": 0.0122, |
| "reward": 0.7857143133878708, |
| "reward_std": 0.05215509608387947, |
| "rewards/unified_reward_func": 0.7857143133878708, |
| "step": 123 |
| }, |
| { |
| "clip_ratio": 0.0006422129372367635, |
| "epoch": 20.551724137931036, |
| "grad_norm": 0.4042130771183457, |
| "kl": 0.53564453125, |
| "learning_rate": 1e-06, |
| "loss": 0.0115, |
| "step": 124 |
| }, |
| { |
| "batch_accuracy": 0.75, |
| "clip_ratio": 0.0, |
| "completion_length": 243.55358123779297, |
| "epoch": 20.689655172413794, |
| "grad_norm": 0.4493382275129834, |
| "kl": 0.3251953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0095, |
| "reward": 0.7500000447034836, |
| "reward_std": 0.04764330945909023, |
| "rewards/unified_reward_func": 0.7500000447034836, |
| "step": 125 |
| }, |
| { |
| "clip_ratio": 0.0014584372693207115, |
| "epoch": 20.82758620689655, |
| "grad_norm": 0.354450124466521, |
| "kl": 0.35595703125, |
| "learning_rate": 1e-06, |
| "loss": 0.009, |
| "step": 126 |
| }, |
| { |
| "batch_accuracy": 0.7589285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 261.5982208251953, |
| "epoch": 21.137931034482758, |
| "grad_norm": 1.1212537411147512, |
| "kl": 0.7021484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0226, |
| "reward": 0.7589285969734192, |
| "reward_std": 0.08070831745862961, |
| "rewards/unified_reward_func": 0.7589285969734192, |
| "step": 127 |
| }, |
| { |
| "clip_ratio": 0.0021403833816293627, |
| "epoch": 21.275862068965516, |
| "grad_norm": 0.564444364760896, |
| "kl": 0.61328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0222, |
| "step": 128 |
| }, |
| { |
| "batch_accuracy": 0.7991071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 283.0089416503906, |
| "epoch": 21.413793103448278, |
| "grad_norm": 1.1428331052805576, |
| "kl": 1.09521484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0133, |
| "reward": 0.7991071790456772, |
| "reward_std": 0.0689915269613266, |
| "rewards/unified_reward_func": 0.7991071790456772, |
| "step": 129 |
| }, |
| { |
| "clip_ratio": 0.0020886963757220656, |
| "epoch": 21.551724137931036, |
| "grad_norm": 1.2791510833897302, |
| "kl": 0.6513671875, |
| "learning_rate": 1e-06, |
| "loss": 0.0122, |
| "step": 130 |
| }, |
| { |
| "batch_accuracy": 0.7544642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 264.71876525878906, |
| "epoch": 21.689655172413794, |
| "grad_norm": 0.5068843628644464, |
| "kl": 0.722900390625, |
| "learning_rate": 1e-06, |
| "loss": -0.0074, |
| "reward": 0.7544643133878708, |
| "reward_std": 0.04824949987232685, |
| "rewards/unified_reward_func": 0.7544643133878708, |
| "step": 131 |
| }, |
| { |
| "clip_ratio": 0.0009302312828367576, |
| "epoch": 21.82758620689655, |
| "grad_norm": 0.7591299641106868, |
| "kl": 0.5947265625, |
| "learning_rate": 1e-06, |
| "loss": -0.008, |
| "step": 132 |
| }, |
| { |
| "batch_accuracy": 0.7455357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 263.6696548461914, |
| "epoch": 22.137931034482758, |
| "grad_norm": 0.9406794669122327, |
| "kl": 0.609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0136, |
| "reward": 0.745535746216774, |
| "reward_std": 0.12249183282256126, |
| "rewards/unified_reward_func": 0.745535746216774, |
| "step": 133 |
| }, |
| { |
| "clip_ratio": 0.0022654080821666867, |
| "epoch": 22.275862068965516, |
| "grad_norm": 4.442348454177717, |
| "kl": 0.37255859375, |
| "learning_rate": 1e-06, |
| "loss": 0.0144, |
| "step": 134 |
| }, |
| { |
| "batch_accuracy": 0.8571428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 261.57590103149414, |
| "epoch": 22.413793103448278, |
| "grad_norm": 1.2706700244569715, |
| "kl": 0.4873046875, |
| "learning_rate": 1e-06, |
| "loss": 0.001, |
| "reward": 0.8571428805589676, |
| "reward_std": 0.08942962251603603, |
| "rewards/unified_reward_func": 0.8571428805589676, |
| "step": 135 |
| }, |
| { |
| "clip_ratio": 0.0022475110308732837, |
| "epoch": 22.551724137931036, |
| "grad_norm": 1.758254727608865, |
| "kl": 0.4248046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0007, |
| "step": 136 |
| }, |
| { |
| "batch_accuracy": 0.7991071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 257.214298248291, |
| "epoch": 22.689655172413794, |
| "grad_norm": 1.9727070210769588, |
| "kl": 0.457763671875, |
| "learning_rate": 1e-06, |
| "loss": 0.0432, |
| "reward": 0.7991071790456772, |
| "reward_std": 0.09619954042136669, |
| "rewards/unified_reward_func": 0.7991071790456772, |
| "step": 137 |
| }, |
| { |
| "clip_ratio": 0.0030386159487534314, |
| "epoch": 22.82758620689655, |
| "grad_norm": 2.897713421787608, |
| "kl": 0.53955078125, |
| "learning_rate": 1e-06, |
| "loss": 0.0428, |
| "step": 138 |
| }, |
| { |
| "batch_accuracy": 0.7678571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 293.4107322692871, |
| "epoch": 23.137931034482758, |
| "grad_norm": 3.4310510361220166, |
| "kl": 1.0546875, |
| "learning_rate": 1e-06, |
| "loss": 0.0137, |
| "reward": 0.7678571864962578, |
| "reward_std": 0.08868780359625816, |
| "rewards/unified_reward_func": 0.7678571864962578, |
| "step": 139 |
| }, |
| { |
| "clip_ratio": 0.0015242814115481451, |
| "epoch": 23.275862068965516, |
| "grad_norm": 1.4261956111254026, |
| "kl": 0.823486328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0134, |
| "step": 140 |
| }, |
| { |
| "batch_accuracy": 0.7678571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 301.7500190734863, |
| "epoch": 23.413793103448278, |
| "grad_norm": 1.6514869046993512, |
| "kl": 0.97900390625, |
| "learning_rate": 1e-06, |
| "loss": 0.019, |
| "reward": 0.7678571790456772, |
| "reward_std": 0.08131169900298119, |
| "rewards/unified_reward_func": 0.7678571790456772, |
| "step": 141 |
| }, |
| { |
| "clip_ratio": 0.001672004524152726, |
| "epoch": 23.551724137931036, |
| "grad_norm": 0.5079581036195917, |
| "kl": 0.87353515625, |
| "learning_rate": 1e-06, |
| "loss": 0.0186, |
| "step": 142 |
| }, |
| { |
| "batch_accuracy": 0.84375, |
| "clip_ratio": 0.0, |
| "completion_length": 219.2678680419922, |
| "epoch": 23.689655172413794, |
| "grad_norm": 0.6205850478806725, |
| "kl": 0.560546875, |
| "learning_rate": 1e-06, |
| "loss": 0.0163, |
| "reward": 0.8437500447034836, |
| "reward_std": 0.08131450787186623, |
| "rewards/unified_reward_func": 0.8437500447034836, |
| "step": 143 |
| }, |
| { |
| "clip_ratio": 0.00177483752486296, |
| "epoch": 23.82758620689655, |
| "grad_norm": 0.6514179664692717, |
| "kl": 0.42724609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0156, |
| "step": 144 |
| }, |
| { |
| "batch_accuracy": 0.7544642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 287.60269927978516, |
| "epoch": 24.137931034482758, |
| "grad_norm": 0.4866499958405398, |
| "kl": 0.7421875, |
| "learning_rate": 1e-06, |
| "loss": 0.03, |
| "reward": 0.754464328289032, |
| "reward_std": 0.0673395898193121, |
| "rewards/unified_reward_func": 0.754464328289032, |
| "step": 145 |
| }, |
| { |
| "clip_ratio": 0.0009519409650238231, |
| "epoch": 24.275862068965516, |
| "grad_norm": 0.36625944324763465, |
| "kl": 0.69873046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0294, |
| "step": 146 |
| }, |
| { |
| "batch_accuracy": 0.7678571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 256.92412185668945, |
| "epoch": 24.413793103448278, |
| "grad_norm": 0.9361673166579322, |
| "kl": 0.89892578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0109, |
| "reward": 0.7678571790456772, |
| "reward_std": 0.08326590247452259, |
| "rewards/unified_reward_func": 0.7678571790456772, |
| "step": 147 |
| }, |
| { |
| "clip_ratio": 0.0012340883258730173, |
| "epoch": 24.551724137931036, |
| "grad_norm": 44.342229710309866, |
| "kl": 1.05078125, |
| "learning_rate": 1e-06, |
| "loss": 0.0233, |
| "step": 148 |
| }, |
| { |
| "batch_accuracy": 0.8571428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 260.4464340209961, |
| "epoch": 24.689655172413794, |
| "grad_norm": 32.73262062897448, |
| "kl": 23.962890625, |
| "learning_rate": 1e-06, |
| "loss": 0.0072, |
| "reward": 0.85714291036129, |
| "reward_std": 0.09198721311986446, |
| "rewards/unified_reward_func": 0.85714291036129, |
| "step": 149 |
| }, |
| { |
| "epoch": 24.82758620689655, |
| "grad_norm": 1.2688282072036752, |
| "learning_rate": 1e-06, |
| "loss": -0.014, |
| "step": 150 |
| }, |
| { |
| "epoch": 24.82758620689655, |
| "eval_batch_accuracy": 0.7154761904761905, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 273.66440734863284, |
| "eval_kl": 1.7647135416666666, |
| "eval_loss": 0.011888084933161736, |
| "eval_reward": 0.7154762188593546, |
| "eval_reward_std": 0.11879824002583822, |
| "eval_rewards/unified_reward_func": 0.7154762188593546, |
| "eval_runtime": 345.8717, |
| "eval_samples_per_second": 0.289, |
| "eval_steps_per_second": 0.006, |
| "step": 150 |
| }, |
| { |
| "batch_accuracy": 0.8571428571428571, |
| "clip_ratio": 0.0010695938399294391, |
| "completion_length": 236.62500762939453, |
| "epoch": 25.137931034482758, |
| "grad_norm": 153.65426313867775, |
| "kl": 13.923583984375, |
| "learning_rate": 1e-06, |
| "loss": 0.0248, |
| "reward": 0.8571428954601288, |
| "reward_std": 0.08613021858036518, |
| "rewards/unified_reward_func": 0.8571428954601288, |
| "step": 151 |
| }, |
| { |
| "clip_ratio": 0.0018477260309737176, |
| "epoch": 25.275862068965516, |
| "grad_norm": 23040.052568690135, |
| "kl": 0.59716796875, |
| "learning_rate": 1e-06, |
| "loss": 7.2765, |
| "step": 152 |
| }, |
| { |
| "batch_accuracy": 0.7366071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 263.20983123779297, |
| "epoch": 25.413793103448278, |
| "grad_norm": 0.4116680840750752, |
| "kl": 0.911865234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0034, |
| "reward": 0.7366071790456772, |
| "reward_std": 0.05441322550177574, |
| "rewards/unified_reward_func": 0.7366071790456772, |
| "step": 153 |
| }, |
| { |
| "clip_ratio": 0.0011999079579254612, |
| "epoch": 25.551724137931036, |
| "grad_norm": 0.3576925903614138, |
| "kl": 0.780517578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0028, |
| "step": 154 |
| }, |
| { |
| "batch_accuracy": 0.7723214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 290.7901954650879, |
| "epoch": 25.689655172413794, |
| "grad_norm": 0.467227142527021, |
| "kl": 0.9345703125, |
| "learning_rate": 1e-06, |
| "loss": 0.0079, |
| "reward": 0.7723214477300644, |
| "reward_std": 0.06612720899283886, |
| "rewards/unified_reward_func": 0.7723214477300644, |
| "step": 155 |
| }, |
| { |
| "clip_ratio": 0.0014949928154237568, |
| "epoch": 25.82758620689655, |
| "grad_norm": 4.261661247612953, |
| "kl": 0.7138671875, |
| "learning_rate": 1e-06, |
| "loss": 0.0073, |
| "step": 156 |
| }, |
| { |
| "batch_accuracy": 0.8660714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 240.62054443359375, |
| "epoch": 26.137931034482758, |
| "grad_norm": 14.49308916720133, |
| "kl": 4.377197265625, |
| "learning_rate": 1e-06, |
| "loss": 0.0017, |
| "reward": 0.8660714775323868, |
| "reward_std": 0.03562259301543236, |
| "rewards/unified_reward_func": 0.8660714775323868, |
| "step": 157 |
| }, |
| { |
| "clip_ratio": 0.0007656059751752764, |
| "epoch": 26.275862068965516, |
| "grad_norm": 0.34782130929462224, |
| "kl": 0.405517578125, |
| "learning_rate": 1e-06, |
| "loss": -0.0021, |
| "step": 158 |
| }, |
| { |
| "batch_accuracy": 0.8080357142857142, |
| "clip_ratio": 0.0, |
| "completion_length": 284.0803756713867, |
| "epoch": 26.413793103448278, |
| "grad_norm": 0.853993640481723, |
| "kl": 0.525390625, |
| "learning_rate": 1e-06, |
| "loss": 0.0448, |
| "reward": 0.808035746216774, |
| "reward_std": 0.12731034867465496, |
| "rewards/unified_reward_func": 0.808035746216774, |
| "step": 159 |
| }, |
| { |
| "clip_ratio": 0.0020131226046942174, |
| "epoch": 26.551724137931036, |
| "grad_norm": 0.6286474424499364, |
| "kl": 0.51904296875, |
| "learning_rate": 1e-06, |
| "loss": 0.0435, |
| "step": 160 |
| }, |
| { |
| "batch_accuracy": 0.7901785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 237.5357322692871, |
| "epoch": 26.689655172413794, |
| "grad_norm": 0.8611827252260262, |
| "kl": 0.66796875, |
| "learning_rate": 1e-06, |
| "loss": 0.0104, |
| "reward": 0.7901785969734192, |
| "reward_std": 0.07094572857022285, |
| "rewards/unified_reward_func": 0.7901785969734192, |
| "step": 161 |
| }, |
| { |
| "clip_ratio": 0.001519633864518255, |
| "epoch": 26.82758620689655, |
| "grad_norm": 0.44506417251011193, |
| "kl": 0.4794921875, |
| "learning_rate": 1e-06, |
| "loss": 0.0093, |
| "step": 162 |
| }, |
| { |
| "batch_accuracy": 0.7410714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 270.714298248291, |
| "epoch": 27.137931034482758, |
| "grad_norm": 0.6713248005427243, |
| "kl": 0.466796875, |
| "learning_rate": 1e-06, |
| "loss": 0.0146, |
| "reward": 0.7410714626312256, |
| "reward_std": 0.07875411584973335, |
| "rewards/unified_reward_func": 0.7410714626312256, |
| "step": 163 |
| }, |
| { |
| "clip_ratio": 0.001083969953469932, |
| "epoch": 27.275862068965516, |
| "grad_norm": 0.4707943331917379, |
| "kl": 0.5361328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0135, |
| "step": 164 |
| }, |
| { |
| "batch_accuracy": 0.7901785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 239.80805206298828, |
| "epoch": 27.413793103448278, |
| "grad_norm": 12.949618265843812, |
| "kl": 12.4248046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0007, |
| "reward": 0.790178582072258, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.790178582072258, |
| "step": 165 |
| }, |
| { |
| "clip_ratio": 0.0008150502981152385, |
| "epoch": 27.551724137931036, |
| "grad_norm": 1.2218985017147648, |
| "kl": 1.4560546875, |
| "learning_rate": 1e-06, |
| "loss": -0.0102, |
| "step": 166 |
| }, |
| { |
| "batch_accuracy": 0.90625, |
| "clip_ratio": 0.0, |
| "completion_length": 237.28572463989258, |
| "epoch": 27.689655172413794, |
| "grad_norm": 0.336672440220934, |
| "kl": 0.643798828125, |
| "learning_rate": 1e-06, |
| "loss": -0.0004, |
| "reward": 0.9062500447034836, |
| "reward_std": 0.03501640260219574, |
| "rewards/unified_reward_func": 0.9062500447034836, |
| "step": 167 |
| }, |
| { |
| "clip_ratio": 0.0004734238755190745, |
| "epoch": 27.82758620689655, |
| "grad_norm": 0.2530434744777561, |
| "kl": 0.53564453125, |
| "learning_rate": 1e-06, |
| "loss": -0.0009, |
| "step": 168 |
| }, |
| { |
| "batch_accuracy": 0.8705357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 239.7187614440918, |
| "epoch": 28.137931034482758, |
| "grad_norm": 0.43116339598093434, |
| "kl": 0.525390625, |
| "learning_rate": 1e-06, |
| "loss": 0.0053, |
| "reward": 0.870535746216774, |
| "reward_std": 0.05441322363913059, |
| "rewards/unified_reward_func": 0.870535746216774, |
| "step": 169 |
| }, |
| { |
| "clip_ratio": 0.0006829075573477894, |
| "epoch": 28.275862068965516, |
| "grad_norm": 0.2890658345945106, |
| "kl": 0.48779296875, |
| "learning_rate": 1e-06, |
| "loss": 0.0046, |
| "step": 170 |
| }, |
| { |
| "batch_accuracy": 0.7142857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 253.49108123779297, |
| "epoch": 28.413793103448278, |
| "grad_norm": 1.594006469856142, |
| "kl": 1.815673828125, |
| "learning_rate": 1e-06, |
| "loss": 0.0126, |
| "reward": 0.7142857611179352, |
| "reward_std": 0.05215509608387947, |
| "rewards/unified_reward_func": 0.7142857611179352, |
| "step": 171 |
| }, |
| { |
| "clip_ratio": 0.0009787115559447557, |
| "epoch": 28.551724137931036, |
| "grad_norm": 0.43736705326513975, |
| "kl": 0.75537109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0114, |
| "step": 172 |
| }, |
| { |
| "batch_accuracy": 0.8616071428571429, |
| "clip_ratio": 0.0, |
| "completion_length": 306.3705520629883, |
| "epoch": 28.689655172413794, |
| "grad_norm": 5.999880286356358, |
| "kl": 3.625, |
| "learning_rate": 1e-06, |
| "loss": 0.0463, |
| "reward": 0.8616071790456772, |
| "reward_std": 0.11784721724689007, |
| "rewards/unified_reward_func": 0.8616071790456772, |
| "step": 173 |
| }, |
| { |
| "clip_ratio": 0.0030182256596162915, |
| "epoch": 28.82758620689655, |
| "grad_norm": 0.809363029282151, |
| "kl": 1.0537109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0438, |
| "step": 174 |
| }, |
| { |
| "batch_accuracy": 0.7321428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 260.3750114440918, |
| "epoch": 29.137931034482758, |
| "grad_norm": 0.5032951734651111, |
| "kl": 0.41552734375, |
| "learning_rate": 1e-06, |
| "loss": -0.0038, |
| "reward": 0.73214291036129, |
| "reward_std": 0.0754547119140625, |
| "rewards/unified_reward_func": 0.73214291036129, |
| "step": 175 |
| }, |
| { |
| "clip_ratio": 0.0010940766078419983, |
| "epoch": 29.275862068965516, |
| "grad_norm": 0.3812137479874885, |
| "kl": 0.50537109375, |
| "learning_rate": 1e-06, |
| "loss": -0.0044, |
| "step": 176 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 297.16518783569336, |
| "epoch": 29.413793103448278, |
| "grad_norm": 0.9027143472678566, |
| "kl": 2.0537109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0154, |
| "reward": 0.848214328289032, |
| "reward_std": 0.11272924393415451, |
| "rewards/unified_reward_func": 0.848214328289032, |
| "step": 177 |
| }, |
| { |
| "clip_ratio": 0.0022016192961018533, |
| "epoch": 29.551724137931036, |
| "grad_norm": 0.655278520928742, |
| "kl": 1.58935546875, |
| "learning_rate": 1e-06, |
| "loss": 0.0141, |
| "step": 178 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 245.64286422729492, |
| "epoch": 29.689655172413794, |
| "grad_norm": 0.6982969760632683, |
| "kl": 0.9599609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0229, |
| "reward": 0.848214328289032, |
| "reward_std": 0.05050762742757797, |
| "rewards/unified_reward_func": 0.848214328289032, |
| "step": 179 |
| }, |
| { |
| "clip_ratio": 0.0013622870174003765, |
| "epoch": 29.82758620689655, |
| "grad_norm": 0.521585705506308, |
| "kl": 0.884765625, |
| "learning_rate": 1e-06, |
| "loss": 0.0221, |
| "step": 180 |
| }, |
| { |
| "batch_accuracy": 0.8080357142857142, |
| "clip_ratio": 0.0, |
| "completion_length": 283.63393783569336, |
| "epoch": 30.137931034482758, |
| "grad_norm": 0.6987516623234656, |
| "kl": 0.95458984375, |
| "learning_rate": 1e-06, |
| "loss": 0.0015, |
| "reward": 0.8080357611179352, |
| "reward_std": 0.070945730432868, |
| "rewards/unified_reward_func": 0.8080357611179352, |
| "step": 181 |
| }, |
| { |
| "clip_ratio": 0.001272890789550729, |
| "epoch": 30.275862068965516, |
| "grad_norm": 1.2679452071601345, |
| "kl": 0.7724609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0012, |
| "step": 182 |
| }, |
| { |
| "batch_accuracy": 0.8705357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 216.00000762939453, |
| "epoch": 30.413793103448278, |
| "grad_norm": 0.23636473762762067, |
| "kl": 0.317626953125, |
| "learning_rate": 1e-06, |
| "loss": -0.0025, |
| "reward": 0.8705357611179352, |
| "reward_std": 0.018483899533748627, |
| "rewards/unified_reward_func": 0.8705357611179352, |
| "step": 183 |
| }, |
| { |
| "clip_ratio": 0.00020213250536471605, |
| "epoch": 30.551724137931036, |
| "grad_norm": 0.2376975480219316, |
| "kl": 0.305419921875, |
| "learning_rate": 1e-06, |
| "loss": -0.0028, |
| "step": 184 |
| }, |
| { |
| "batch_accuracy": 0.7946428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 237.61608123779297, |
| "epoch": 30.689655172413794, |
| "grad_norm": 0.9304541510624866, |
| "kl": 1.8798828125, |
| "learning_rate": 1e-06, |
| "loss": 0.0128, |
| "reward": 0.7946428954601288, |
| "reward_std": 0.056364621967077255, |
| "rewards/unified_reward_func": 0.7946428954601288, |
| "step": 185 |
| }, |
| { |
| "clip_ratio": 0.001274522306630388, |
| "epoch": 30.82758620689655, |
| "grad_norm": 0.6160668926137608, |
| "kl": 1.09228515625, |
| "learning_rate": 1e-06, |
| "loss": 0.012, |
| "step": 186 |
| }, |
| { |
| "batch_accuracy": 0.7857142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 264.026798248291, |
| "epoch": 31.137931034482758, |
| "grad_norm": 0.8381457440700479, |
| "kl": 0.68115234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0165, |
| "reward": 0.7857143133878708, |
| "reward_std": 0.08357263542711735, |
| "rewards/unified_reward_func": 0.7857143133878708, |
| "step": 187 |
| }, |
| { |
| "clip_ratio": 0.0035605556913651526, |
| "epoch": 31.275862068965516, |
| "grad_norm": 1.1856915368204115, |
| "kl": 0.740234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0153, |
| "step": 188 |
| }, |
| { |
| "batch_accuracy": 0.9241071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 252.27679443359375, |
| "epoch": 31.413793103448278, |
| "grad_norm": 0.6661533926688221, |
| "kl": 0.9140625, |
| "learning_rate": 1e-06, |
| "loss": -0.0026, |
| "reward": 0.9241071939468384, |
| "reward_std": 0.0689915269613266, |
| "rewards/unified_reward_func": 0.9241071939468384, |
| "step": 189 |
| }, |
| { |
| "clip_ratio": 0.0012773344933521003, |
| "epoch": 31.551724137931036, |
| "grad_norm": 0.9096020542402118, |
| "kl": 0.81103515625, |
| "learning_rate": 1e-06, |
| "loss": -0.0031, |
| "step": 190 |
| }, |
| { |
| "batch_accuracy": 0.75, |
| "clip_ratio": 0.0, |
| "completion_length": 231.93750762939453, |
| "epoch": 31.689655172413794, |
| "grad_norm": 0.5985491059230904, |
| "kl": 1.140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0038, |
| "reward": 0.7500000447034836, |
| "reward_std": 0.08070831745862961, |
| "rewards/unified_reward_func": 0.7500000447034836, |
| "step": 191 |
| }, |
| { |
| "clip_ratio": 0.001351288432488218, |
| "epoch": 31.82758620689655, |
| "grad_norm": 0.43023596336510533, |
| "kl": 1.08935546875, |
| "learning_rate": 1e-06, |
| "loss": 0.0028, |
| "step": 192 |
| }, |
| { |
| "batch_accuracy": 0.8125, |
| "clip_ratio": 0.0, |
| "completion_length": 250.65179443359375, |
| "epoch": 32.13793103448276, |
| "grad_norm": 2.478437026215032, |
| "kl": 2.31494140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0045, |
| "reward": 0.8125000298023224, |
| "reward_std": 0.05050762556493282, |
| "rewards/unified_reward_func": 0.8125000298023224, |
| "step": 193 |
| }, |
| { |
| "clip_ratio": 0.0017007194110192358, |
| "epoch": 32.275862068965516, |
| "grad_norm": 77.17716361138062, |
| "kl": 1.08447265625, |
| "learning_rate": 1e-06, |
| "loss": 0.0218, |
| "step": 194 |
| }, |
| { |
| "batch_accuracy": 0.8303571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 234.59376525878906, |
| "epoch": 32.41379310344828, |
| "grad_norm": 6.966001947932676, |
| "kl": 1.583984375, |
| "learning_rate": 1e-06, |
| "loss": 0.0106, |
| "reward": 0.8303571790456772, |
| "reward_std": 0.04959750920534134, |
| "rewards/unified_reward_func": 0.8303571790456772, |
| "step": 195 |
| }, |
| { |
| "clip_ratio": 0.0010828501835931093, |
| "epoch": 32.55172413793103, |
| "grad_norm": 0.41210893717986125, |
| "kl": 0.559326171875, |
| "learning_rate": 1e-06, |
| "loss": 0.0097, |
| "step": 196 |
| }, |
| { |
| "batch_accuracy": 0.8705357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 238.99108123779297, |
| "epoch": 32.689655172413794, |
| "grad_norm": 0.5727595579664349, |
| "kl": 0.84423828125, |
| "learning_rate": 1e-06, |
| "loss": 0.0, |
| "reward": 0.870535746216774, |
| "reward_std": 0.04569191299378872, |
| "rewards/unified_reward_func": 0.870535746216774, |
| "step": 197 |
| }, |
| { |
| "clip_ratio": 0.0004849687684327364, |
| "epoch": 32.827586206896555, |
| "grad_norm": 0.34508485808239964, |
| "kl": 0.692138671875, |
| "learning_rate": 1e-06, |
| "loss": -0.0008, |
| "step": 198 |
| }, |
| { |
| "batch_accuracy": 0.8660714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 233.93750762939453, |
| "epoch": 33.13793103448276, |
| "grad_norm": 0.6389097363094803, |
| "kl": 0.79345703125, |
| "learning_rate": 1e-06, |
| "loss": 0.0107, |
| "reward": 0.8660714775323868, |
| "reward_std": 0.056364623829722404, |
| "rewards/unified_reward_func": 0.8660714775323868, |
| "step": 199 |
| }, |
| { |
| "epoch": 33.275862068965516, |
| "grad_norm": 0.4172479026848871, |
| "learning_rate": 1e-06, |
| "loss": 0.0098, |
| "step": 200 |
| }, |
| { |
| "epoch": 33.275862068965516, |
| "eval_batch_accuracy": 0.7095238095238094, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 253.85001017252605, |
| "eval_kl": 0.13782552083333333, |
| "eval_loss": 0.012882467359304428, |
| "eval_reward": 0.7095238447189331, |
| "eval_reward_std": 0.14272718926270803, |
| "eval_rewards/unified_reward_func": 0.7095238447189331, |
| "eval_runtime": 309.75, |
| "eval_samples_per_second": 0.323, |
| "eval_steps_per_second": 0.006, |
| "step": 200 |
| }, |
| { |
| "batch_accuracy": 0.8125, |
| "clip_ratio": 0.00047189262841129676, |
| "completion_length": 216.61608123779297, |
| "epoch": 33.41379310344828, |
| "grad_norm": 0.34075034379789443, |
| "kl": 0.3455810546875, |
| "learning_rate": 1e-06, |
| "loss": 0.0028, |
| "reward": 0.8125000298023224, |
| "reward_std": 0.0417863167822361, |
| "rewards/unified_reward_func": 0.8125000298023224, |
| "step": 201 |
| }, |
| { |
| "clip_ratio": 0.0005730669654440135, |
| "epoch": 33.55172413793103, |
| "grad_norm": 0.24516084654318515, |
| "kl": 0.108154296875, |
| "learning_rate": 1e-06, |
| "loss": 0.0023, |
| "step": 202 |
| }, |
| { |
| "batch_accuracy": 0.8258928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 247.77679443359375, |
| "epoch": 33.689655172413794, |
| "grad_norm": 0.7986333842953461, |
| "kl": 0.114990234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0313, |
| "reward": 0.8258928805589676, |
| "reward_std": 0.060270216315984726, |
| "rewards/unified_reward_func": 0.8258928805589676, |
| "step": 203 |
| }, |
| { |
| "clip_ratio": 0.0017704838537611067, |
| "epoch": 33.827586206896555, |
| "grad_norm": 0.5264166409776956, |
| "kl": 0.1175537109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0304, |
| "step": 204 |
| }, |
| { |
| "batch_accuracy": 0.7366071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 239.7544822692871, |
| "epoch": 34.13793103448276, |
| "grad_norm": 0.5322234496834484, |
| "kl": 0.1153564453125, |
| "learning_rate": 1e-06, |
| "loss": -0.0007, |
| "reward": 0.736607164144516, |
| "reward_std": 0.07606089860200882, |
| "rewards/unified_reward_func": 0.736607164144516, |
| "step": 205 |
| }, |
| { |
| "clip_ratio": 0.0016890191909624264, |
| "epoch": 34.275862068965516, |
| "grad_norm": 0.37722043952212947, |
| "kl": 0.1268310546875, |
| "learning_rate": 1e-06, |
| "loss": -0.0017, |
| "step": 206 |
| }, |
| { |
| "batch_accuracy": 0.9017857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 260.91965103149414, |
| "epoch": 34.41379310344828, |
| "grad_norm": 0.6213850315224072, |
| "kl": 0.1051025390625, |
| "learning_rate": 1e-06, |
| "loss": 0.0085, |
| "reward": 0.9017857313156128, |
| "reward_std": 0.07576144114136696, |
| "rewards/unified_reward_func": 0.9017857313156128, |
| "step": 207 |
| }, |
| { |
| "clip_ratio": 0.00208102646865882, |
| "epoch": 34.55172413793103, |
| "grad_norm": 0.41446319845532803, |
| "kl": 0.11376953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0079, |
| "step": 208 |
| }, |
| { |
| "batch_accuracy": 0.8080357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 220.27679443359375, |
| "epoch": 34.689655172413794, |
| "grad_norm": 0.34395769515755015, |
| "kl": 0.3416748046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0033, |
| "reward": 0.8080357313156128, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.8080357313156128, |
| "step": 209 |
| }, |
| { |
| "clip_ratio": 0.0005923433491261676, |
| "epoch": 34.827586206896555, |
| "grad_norm": 0.24381267530999462, |
| "kl": 0.303466796875, |
| "learning_rate": 1e-06, |
| "loss": 0.0028, |
| "step": 210 |
| }, |
| { |
| "batch_accuracy": 0.8303571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 201.46429824829102, |
| "epoch": 35.13793103448276, |
| "grad_norm": 0.5610912818294096, |
| "kl": 0.1767578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0062, |
| "reward": 0.8303571790456772, |
| "reward_std": 0.04764331132173538, |
| "rewards/unified_reward_func": 0.8303571790456772, |
| "step": 211 |
| }, |
| { |
| "clip_ratio": 0.002142434357665479, |
| "epoch": 35.275862068965516, |
| "grad_norm": 12.075891369202603, |
| "kl": 0.276123046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0058, |
| "step": 212 |
| }, |
| { |
| "batch_accuracy": 0.71875, |
| "clip_ratio": 0.0, |
| "completion_length": 267.88393783569336, |
| "epoch": 35.41379310344828, |
| "grad_norm": 4.675264026045327, |
| "kl": 0.266845703125, |
| "learning_rate": 1e-06, |
| "loss": 0.0227, |
| "reward": 0.7187500447034836, |
| "reward_std": 0.060270220041275024, |
| "rewards/unified_reward_func": 0.7187500447034836, |
| "step": 213 |
| }, |
| { |
| "clip_ratio": 0.001382013550028205, |
| "epoch": 35.55172413793103, |
| "grad_norm": 3.166666135840778, |
| "kl": 0.205322265625, |
| "learning_rate": 1e-06, |
| "loss": 0.023, |
| "step": 214 |
| }, |
| { |
| "batch_accuracy": 0.8526785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 223.82143783569336, |
| "epoch": 35.689655172413794, |
| "grad_norm": 0.5784662868247091, |
| "kl": 0.2890625, |
| "learning_rate": 1e-06, |
| "loss": 0.0089, |
| "reward": 0.8526786118745804, |
| "reward_std": 0.08656250685453415, |
| "rewards/unified_reward_func": 0.8526786118745804, |
| "step": 215 |
| }, |
| { |
| "clip_ratio": 0.0020873736939392984, |
| "epoch": 35.827586206896555, |
| "grad_norm": 0.5514140064721655, |
| "kl": 0.25439453125, |
| "learning_rate": 1e-06, |
| "loss": 0.0083, |
| "step": 216 |
| }, |
| { |
| "batch_accuracy": 0.84375, |
| "clip_ratio": 0.0, |
| "completion_length": 216.41072463989258, |
| "epoch": 36.13793103448276, |
| "grad_norm": 0.5359392827868366, |
| "kl": 0.249267578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0048, |
| "reward": 0.8437500298023224, |
| "reward_std": 0.05441322363913059, |
| "rewards/unified_reward_func": 0.8437500298023224, |
| "step": 217 |
| }, |
| { |
| "clip_ratio": 0.0010411773982923478, |
| "epoch": 36.275862068965516, |
| "grad_norm": 0.4041522386212472, |
| "kl": 0.2666015625, |
| "learning_rate": 1e-06, |
| "loss": 0.0042, |
| "step": 218 |
| }, |
| { |
| "batch_accuracy": 0.7678571428571429, |
| "clip_ratio": 0.0, |
| "completion_length": 235.70536422729492, |
| "epoch": 36.41379310344828, |
| "grad_norm": 59.360952385482854, |
| "kl": 1.32275390625, |
| "learning_rate": 1e-06, |
| "loss": -0.0075, |
| "reward": 0.7678571790456772, |
| "reward_std": 0.060876404866576195, |
| "rewards/unified_reward_func": 0.7678571790456772, |
| "step": 219 |
| }, |
| { |
| "clip_ratio": 0.003026856400538236, |
| "epoch": 36.55172413793103, |
| "grad_norm": 198.16594895250253, |
| "kl": 3.1953125, |
| "learning_rate": 1e-06, |
| "loss": -0.0052, |
| "step": 220 |
| }, |
| { |
| "batch_accuracy": 0.8080357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 215.12947463989258, |
| "epoch": 36.689655172413794, |
| "grad_norm": 1.110727771046951, |
| "kl": 0.7587890625, |
| "learning_rate": 1e-06, |
| "loss": 0.0131, |
| "reward": 0.8080357611179352, |
| "reward_std": 0.11377052217721939, |
| "rewards/unified_reward_func": 0.8080357611179352, |
| "step": 221 |
| }, |
| { |
| "clip_ratio": 0.0022536990436492488, |
| "epoch": 36.827586206896555, |
| "grad_norm": 1.226956128980916, |
| "kl": 0.72900390625, |
| "learning_rate": 1e-06, |
| "loss": 0.012, |
| "step": 222 |
| }, |
| { |
| "batch_accuracy": 0.8616071428571429, |
| "clip_ratio": 0.0, |
| "completion_length": 211.2812614440918, |
| "epoch": 37.13793103448276, |
| "grad_norm": 1.2105264518287366, |
| "kl": 0.985107421875, |
| "learning_rate": 1e-06, |
| "loss": 0.0147, |
| "reward": 0.861607164144516, |
| "reward_std": 0.056970810517668724, |
| "rewards/unified_reward_func": 0.861607164144516, |
| "step": 223 |
| }, |
| { |
| "clip_ratio": 0.001337285284535028, |
| "epoch": 37.275862068965516, |
| "grad_norm": 6.096977977454511, |
| "kl": 0.493896484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0163, |
| "step": 224 |
| }, |
| { |
| "batch_accuracy": 0.9017857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 209.11608123779297, |
| "epoch": 37.41379310344828, |
| "grad_norm": 1.062706817262316, |
| "kl": 1.128662109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0079, |
| "reward": 0.9017857611179352, |
| "reward_std": 0.04764330945909023, |
| "rewards/unified_reward_func": 0.9017857611179352, |
| "step": 225 |
| }, |
| { |
| "clip_ratio": 0.0014827463019173592, |
| "epoch": 37.55172413793103, |
| "grad_norm": 0.4654980038459283, |
| "kl": 0.870361328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0071, |
| "step": 226 |
| }, |
| { |
| "batch_accuracy": 0.7991071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 204.71429443359375, |
| "epoch": 37.689655172413794, |
| "grad_norm": 2.0275758552802943, |
| "kl": 1.9033203125, |
| "learning_rate": 1e-06, |
| "loss": 0.011, |
| "reward": 0.799107164144516, |
| "reward_std": 0.05441322736442089, |
| "rewards/unified_reward_func": 0.799107164144516, |
| "step": 227 |
| }, |
| { |
| "clip_ratio": 0.0018361783877480775, |
| "epoch": 37.827586206896555, |
| "grad_norm": 0.6875113705883846, |
| "kl": 0.61181640625, |
| "learning_rate": 1e-06, |
| "loss": 0.0093, |
| "step": 228 |
| }, |
| { |
| "batch_accuracy": 0.9017857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 224.85268783569336, |
| "epoch": 38.13793103448276, |
| "grad_norm": 5.487262077964094, |
| "kl": 2.03857421875, |
| "learning_rate": 1e-06, |
| "loss": 0.0024, |
| "reward": 0.9017857313156128, |
| "reward_std": 0.04764331132173538, |
| "rewards/unified_reward_func": 0.9017857313156128, |
| "step": 229 |
| }, |
| { |
| "clip_ratio": 0.0015996035072021186, |
| "epoch": 38.275862068965516, |
| "grad_norm": 1450.0954313928476, |
| "kl": 0.55322265625, |
| "learning_rate": 1e-06, |
| "loss": 1.3522, |
| "step": 230 |
| }, |
| { |
| "batch_accuracy": 0.7857142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 225.71429443359375, |
| "epoch": 38.41379310344828, |
| "grad_norm": 0.6598845880184323, |
| "kl": 0.759765625, |
| "learning_rate": 1e-06, |
| "loss": -0.0111, |
| "reward": 0.785714328289032, |
| "reward_std": 0.0641758143901825, |
| "rewards/unified_reward_func": 0.785714328289032, |
| "step": 231 |
| }, |
| { |
| "clip_ratio": 0.001130485899921041, |
| "epoch": 38.55172413793103, |
| "grad_norm": 0.6240844125235968, |
| "kl": 0.95556640625, |
| "learning_rate": 1e-06, |
| "loss": -0.0118, |
| "step": 232 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 202.25447463989258, |
| "epoch": 38.689655172413794, |
| "grad_norm": 0.6091800162021634, |
| "kl": 0.791015625, |
| "learning_rate": 1e-06, |
| "loss": 0.0005, |
| "reward": 0.8839285969734192, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839285969734192, |
| "step": 233 |
| }, |
| { |
| "clip_ratio": 0.0003748063463717699, |
| "epoch": 38.827586206896555, |
| "grad_norm": 0.2670190645611779, |
| "kl": 0.51318359375, |
| "learning_rate": 1e-06, |
| "loss": -0.0002, |
| "step": 234 |
| }, |
| { |
| "batch_accuracy": 0.8705357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 197.33483123779297, |
| "epoch": 39.13793103448276, |
| "grad_norm": 8.036094604397608, |
| "kl": 6.68310546875, |
| "learning_rate": 1e-06, |
| "loss": 0.0196, |
| "reward": 0.870535746216774, |
| "reward_std": 0.04373771324753761, |
| "rewards/unified_reward_func": 0.870535746216774, |
| "step": 235 |
| }, |
| { |
| "clip_ratio": 0.0014344880473800004, |
| "epoch": 39.275862068965516, |
| "grad_norm": 1.3381116842000897, |
| "kl": 2.285400390625, |
| "learning_rate": 1e-06, |
| "loss": 0.0158, |
| "step": 236 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 246.4419708251953, |
| "epoch": 39.41379310344828, |
| "grad_norm": 0.7703391548316393, |
| "kl": 0.4609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0111, |
| "reward": 0.8482143133878708, |
| "reward_std": 0.0774089116603136, |
| "rewards/unified_reward_func": 0.8482143133878708, |
| "step": 237 |
| }, |
| { |
| "clip_ratio": 0.0026967041776515543, |
| "epoch": 39.55172413793103, |
| "grad_norm": 0.6278183908739686, |
| "kl": 0.439453125, |
| "learning_rate": 1e-06, |
| "loss": 0.0107, |
| "step": 238 |
| }, |
| { |
| "batch_accuracy": 0.875, |
| "clip_ratio": 0.0, |
| "completion_length": 216.03125762939453, |
| "epoch": 39.689655172413794, |
| "grad_norm": 0.486258783542183, |
| "kl": 0.41552734375, |
| "learning_rate": 1e-06, |
| "loss": -0.0024, |
| "reward": 0.8750000298023224, |
| "reward_std": 0.033065006136894226, |
| "rewards/unified_reward_func": 0.8750000298023224, |
| "step": 239 |
| }, |
| { |
| "clip_ratio": 0.0005882278346689418, |
| "epoch": 39.827586206896555, |
| "grad_norm": 1.053399436053242, |
| "kl": 0.277587890625, |
| "learning_rate": 1e-06, |
| "loss": -0.0023, |
| "step": 240 |
| }, |
| { |
| "batch_accuracy": 0.8526785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 231.79019165039062, |
| "epoch": 40.13793103448276, |
| "grad_norm": 0.7411372005958455, |
| "kl": 0.4443359375, |
| "learning_rate": 1e-06, |
| "loss": 0.0114, |
| "reward": 0.8526785969734192, |
| "reward_std": 0.05441322363913059, |
| "rewards/unified_reward_func": 0.8526785969734192, |
| "step": 241 |
| }, |
| { |
| "clip_ratio": 0.0011974502704106271, |
| "epoch": 40.275862068965516, |
| "grad_norm": 0.5129553126144167, |
| "kl": 0.422119140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0104, |
| "step": 242 |
| }, |
| { |
| "batch_accuracy": 0.8571428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 215.67858123779297, |
| "epoch": 40.41379310344828, |
| "grad_norm": 0.7342681774537219, |
| "kl": 0.75634765625, |
| "learning_rate": 1e-06, |
| "loss": -0.0114, |
| "reward": 0.8571428805589676, |
| "reward_std": 0.06417581252753735, |
| "rewards/unified_reward_func": 0.8571428805589676, |
| "step": 243 |
| }, |
| { |
| "clip_ratio": 0.00107419173582457, |
| "epoch": 40.55172413793103, |
| "grad_norm": 1.970726406090723, |
| "kl": 0.614013671875, |
| "learning_rate": 1e-06, |
| "loss": -0.0116, |
| "step": 244 |
| }, |
| { |
| "batch_accuracy": 0.8125, |
| "clip_ratio": 0.0, |
| "completion_length": 255.63840103149414, |
| "epoch": 40.689655172413794, |
| "grad_norm": 1.4462939125357268, |
| "kl": 0.83251953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0111, |
| "reward": 0.8125000447034836, |
| "reward_std": 0.05050762742757797, |
| "rewards/unified_reward_func": 0.8125000447034836, |
| "step": 245 |
| }, |
| { |
| "clip_ratio": 0.0021176336449570954, |
| "epoch": 40.827586206896555, |
| "grad_norm": 0.6476801537145345, |
| "kl": 0.7197265625, |
| "learning_rate": 1e-06, |
| "loss": 0.0103, |
| "step": 246 |
| }, |
| { |
| "batch_accuracy": 0.9196428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 205.39733123779297, |
| "epoch": 41.13793103448276, |
| "grad_norm": 0.4995189751066373, |
| "kl": 0.55029296875, |
| "learning_rate": 1e-06, |
| "loss": 0.0013, |
| "reward": 0.9196428954601288, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9196428954601288, |
| "step": 247 |
| }, |
| { |
| "clip_ratio": 0.0004891223652521148, |
| "epoch": 41.275862068965516, |
| "grad_norm": 0.2501727453686479, |
| "kl": 0.30615234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0008, |
| "step": 248 |
| }, |
| { |
| "batch_accuracy": 0.75, |
| "clip_ratio": 0.0, |
| "completion_length": 233.22768783569336, |
| "epoch": 41.41379310344828, |
| "grad_norm": 0.6614211017916019, |
| "kl": 0.5869140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0071, |
| "reward": 0.7500000298023224, |
| "reward_std": 0.053500302135944366, |
| "rewards/unified_reward_func": 0.7500000298023224, |
| "step": 249 |
| }, |
| { |
| "epoch": 41.55172413793103, |
| "grad_norm": 0.5420823826241697, |
| "learning_rate": 1e-06, |
| "loss": 0.0067, |
| "step": 250 |
| }, |
| { |
| "epoch": 41.55172413793103, |
| "eval_batch_accuracy": 0.7309523809523809, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 242.77955525716146, |
| "eval_kl": 1.0609375, |
| "eval_loss": 0.01743003912270069, |
| "eval_reward": 0.7309524257977803, |
| "eval_reward_std": 0.14946034500996272, |
| "eval_rewards/unified_reward_func": 0.7309524257977803, |
| "eval_runtime": 300.631, |
| "eval_samples_per_second": 0.333, |
| "eval_steps_per_second": 0.007, |
| "step": 250 |
| }, |
| { |
| "batch_accuracy": 0.84375, |
| "clip_ratio": 0.0006140165933175012, |
| "completion_length": 239.43305206298828, |
| "epoch": 41.689655172413794, |
| "grad_norm": 10.36975411089091, |
| "kl": 4.4228515625, |
| "learning_rate": 1e-06, |
| "loss": 0.004, |
| "reward": 0.8437500596046448, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.8437500596046448, |
| "step": 251 |
| }, |
| { |
| "clip_ratio": 0.0005719505425076932, |
| "epoch": 41.827586206896555, |
| "grad_norm": 1.5376205862711856, |
| "kl": 0.7177734375, |
| "learning_rate": 1e-06, |
| "loss": -0.0016, |
| "step": 252 |
| }, |
| { |
| "batch_accuracy": 0.8794642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 239.58483505249023, |
| "epoch": 42.13793103448276, |
| "grad_norm": 0.4355553650409747, |
| "kl": 0.420166015625, |
| "learning_rate": 1e-06, |
| "loss": -0.004, |
| "reward": 0.8794643133878708, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.8794643133878708, |
| "step": 253 |
| }, |
| { |
| "clip_ratio": 0.0004529447469394654, |
| "epoch": 42.275862068965516, |
| "grad_norm": 0.3473121365753115, |
| "kl": 0.358154296875, |
| "learning_rate": 1e-06, |
| "loss": -0.0045, |
| "step": 254 |
| }, |
| { |
| "batch_accuracy": 0.7901785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 249.67858123779297, |
| "epoch": 42.41379310344828, |
| "grad_norm": 0.955097933412004, |
| "kl": 0.576171875, |
| "learning_rate": 1e-06, |
| "loss": 0.0245, |
| "reward": 0.7901785969734192, |
| "reward_std": 0.07966703735291958, |
| "rewards/unified_reward_func": 0.7901785969734192, |
| "step": 255 |
| }, |
| { |
| "clip_ratio": 0.0023401440121233463, |
| "epoch": 42.55172413793103, |
| "grad_norm": 0.6440396509406432, |
| "kl": 0.64599609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0233, |
| "step": 256 |
| }, |
| { |
| "batch_accuracy": 0.8660714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 203.26340103149414, |
| "epoch": 42.689655172413794, |
| "grad_norm": 0.7450317244478006, |
| "kl": 0.638671875, |
| "learning_rate": 1e-06, |
| "loss": -0.0029, |
| "reward": 0.8660714626312256, |
| "reward_std": 0.06704013422131538, |
| "rewards/unified_reward_func": 0.8660714626312256, |
| "step": 257 |
| }, |
| { |
| "clip_ratio": 0.0013811402022838593, |
| "epoch": 42.827586206896555, |
| "grad_norm": 0.7976763809242704, |
| "kl": 0.36376953125, |
| "learning_rate": 1e-06, |
| "loss": -0.0034, |
| "step": 258 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 236.8303680419922, |
| "epoch": 43.13793103448276, |
| "grad_norm": 0.5496733412364695, |
| "kl": 0.44580078125, |
| "learning_rate": 1e-06, |
| "loss": 0.009, |
| "reward": 0.8482143431901932, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8482143431901932, |
| "step": 259 |
| }, |
| { |
| "clip_ratio": 0.0004358286823844537, |
| "epoch": 43.275862068965516, |
| "grad_norm": 0.4551862846969153, |
| "kl": 0.41845703125, |
| "learning_rate": 1e-06, |
| "loss": 0.0083, |
| "step": 260 |
| }, |
| { |
| "batch_accuracy": 0.8258928571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 219.31251525878906, |
| "epoch": 43.41379310344828, |
| "grad_norm": 1.3480682509696915, |
| "kl": 1.4970703125, |
| "learning_rate": 1e-06, |
| "loss": 0.0263, |
| "reward": 0.8258928805589676, |
| "reward_std": 0.07966704294085503, |
| "rewards/unified_reward_func": 0.8258928805589676, |
| "step": 261 |
| }, |
| { |
| "clip_ratio": 0.0032285154156852514, |
| "epoch": 43.55172413793103, |
| "grad_norm": 1.3108389523737243, |
| "kl": 0.986328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0257, |
| "step": 262 |
| }, |
| { |
| "batch_accuracy": 0.8660714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 250.83483505249023, |
| "epoch": 43.689655172413794, |
| "grad_norm": 1.5574408610060597, |
| "kl": 1.7861328125, |
| "learning_rate": 1e-06, |
| "loss": -0.0084, |
| "reward": 0.8660714626312256, |
| "reward_std": 0.11468344740569592, |
| "rewards/unified_reward_func": 0.8660714626312256, |
| "step": 263 |
| }, |
| { |
| "clip_ratio": 0.005030798492953181, |
| "epoch": 43.827586206896555, |
| "grad_norm": 4.383711031864911, |
| "kl": 1.2841796875, |
| "learning_rate": 1e-06, |
| "loss": -0.0092, |
| "step": 264 |
| }, |
| { |
| "batch_accuracy": 0.8392857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 214.05804824829102, |
| "epoch": 44.13793103448276, |
| "grad_norm": 1.3084821514107963, |
| "kl": 0.7490234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0028, |
| "reward": 0.839285746216774, |
| "reward_std": 0.03111080639064312, |
| "rewards/unified_reward_func": 0.839285746216774, |
| "step": 265 |
| }, |
| { |
| "clip_ratio": 0.0014369665295816958, |
| "epoch": 44.275862068965516, |
| "grad_norm": 4.089177210944676, |
| "kl": 1.0859375, |
| "learning_rate": 1e-06, |
| "loss": 0.0029, |
| "step": 266 |
| }, |
| { |
| "batch_accuracy": 0.9196428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 230.01787185668945, |
| "epoch": 44.41379310344828, |
| "grad_norm": 8.798834640093343, |
| "kl": 5.603515625, |
| "learning_rate": 1e-06, |
| "loss": 0.0153, |
| "reward": 0.9196428954601288, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9196428954601288, |
| "step": 267 |
| }, |
| { |
| "clip_ratio": 0.001320267387200147, |
| "epoch": 44.55172413793103, |
| "grad_norm": 2.2251765859025703, |
| "kl": 2.681640625, |
| "learning_rate": 1e-06, |
| "loss": 0.0129, |
| "step": 268 |
| }, |
| { |
| "batch_accuracy": 0.8616071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 226.36608123779297, |
| "epoch": 44.689655172413794, |
| "grad_norm": 1940.7593108258543, |
| "kl": 1058.919921875, |
| "learning_rate": 1e-06, |
| "loss": 1.0647, |
| "reward": 0.861607164144516, |
| "reward_std": 0.07966703921556473, |
| "rewards/unified_reward_func": 0.861607164144516, |
| "step": 269 |
| }, |
| { |
| "clip_ratio": 0.0029754533316008747, |
| "epoch": 44.827586206896555, |
| "grad_norm": 838.2173848039702, |
| "kl": 5.6171875, |
| "learning_rate": 1e-06, |
| "loss": 0.2984, |
| "step": 270 |
| }, |
| { |
| "batch_accuracy": 0.8348214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 257.1026840209961, |
| "epoch": 45.13793103448276, |
| "grad_norm": 3.3428194911061606, |
| "kl": 2.40625, |
| "learning_rate": 1e-06, |
| "loss": -0.007, |
| "reward": 0.8348214477300644, |
| "reward_std": 0.03171699680387974, |
| "rewards/unified_reward_func": 0.8348214477300644, |
| "step": 271 |
| }, |
| { |
| "clip_ratio": 0.0009366457816213369, |
| "epoch": 45.275862068965516, |
| "grad_norm": 0.6784487520271747, |
| "kl": 1.31005859375, |
| "learning_rate": 1e-06, |
| "loss": -0.0081, |
| "step": 272 |
| }, |
| { |
| "batch_accuracy": 0.8392857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 222.82590866088867, |
| "epoch": 45.41379310344828, |
| "grad_norm": 0.5821113324215218, |
| "kl": 0.99609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0089, |
| "reward": 0.839285746216774, |
| "reward_std": 0.03111080639064312, |
| "rewards/unified_reward_func": 0.839285746216774, |
| "step": 273 |
| }, |
| { |
| "clip_ratio": 0.0006448913627536967, |
| "epoch": 45.55172413793103, |
| "grad_norm": 0.5386933478140308, |
| "kl": 0.67138671875, |
| "learning_rate": 1e-06, |
| "loss": 0.0086, |
| "step": 274 |
| }, |
| { |
| "batch_accuracy": 0.8258928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 240.4196548461914, |
| "epoch": 45.689655172413794, |
| "grad_norm": 2.679436922054985, |
| "kl": 2.19140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0119, |
| "reward": 0.8258928954601288, |
| "reward_std": 0.0709457267075777, |
| "rewards/unified_reward_func": 0.8258928954601288, |
| "step": 275 |
| }, |
| { |
| "clip_ratio": 0.0021198716713115573, |
| "epoch": 45.827586206896555, |
| "grad_norm": 0.7430334891757607, |
| "kl": 1.31494140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0109, |
| "step": 276 |
| }, |
| { |
| "batch_accuracy": 0.8526785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 242.66965103149414, |
| "epoch": 46.13793103448276, |
| "grad_norm": 0.6689116919401598, |
| "kl": 0.66943359375, |
| "learning_rate": 1e-06, |
| "loss": 0.0053, |
| "reward": 0.8526785969734192, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.8526785969734192, |
| "step": 277 |
| }, |
| { |
| "clip_ratio": 0.0001296815025852993, |
| "epoch": 46.275862068965516, |
| "grad_norm": 0.48746993806188527, |
| "kl": 0.54248046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0049, |
| "step": 278 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 247.94644165039062, |
| "epoch": 46.41379310344828, |
| "grad_norm": 0.8025427426991367, |
| "kl": 1.60498046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0044, |
| "reward": 0.8482143431901932, |
| "reward_std": 0.07740890793502331, |
| "rewards/unified_reward_func": 0.8482143431901932, |
| "step": 279 |
| }, |
| { |
| "clip_ratio": 0.0018109494121745229, |
| "epoch": 46.55172413793103, |
| "grad_norm": 0.717943396090161, |
| "kl": 0.910400390625, |
| "learning_rate": 1e-06, |
| "loss": 0.0036, |
| "step": 280 |
| }, |
| { |
| "batch_accuracy": 0.8571428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 220.4687614440918, |
| "epoch": 46.689655172413794, |
| "grad_norm": 1.0984351537808166, |
| "kl": 1.27294921875, |
| "learning_rate": 1e-06, |
| "loss": -0.0036, |
| "reward": 0.85714291036129, |
| "reward_std": 0.050200898200273514, |
| "rewards/unified_reward_func": 0.85714291036129, |
| "step": 281 |
| }, |
| { |
| "clip_ratio": 0.0013987061684019864, |
| "epoch": 46.827586206896555, |
| "grad_norm": 0.8855299850661535, |
| "kl": 0.6982421875, |
| "learning_rate": 1e-06, |
| "loss": -0.004, |
| "step": 282 |
| }, |
| { |
| "batch_accuracy": 0.78125, |
| "clip_ratio": 0.0, |
| "completion_length": 237.95091247558594, |
| "epoch": 47.13793103448276, |
| "grad_norm": 0.7861590293685551, |
| "kl": 0.663330078125, |
| "learning_rate": 1e-06, |
| "loss": 0.0085, |
| "reward": 0.7812500298023224, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.7812500298023224, |
| "step": 283 |
| }, |
| { |
| "clip_ratio": 0.00020263424084987491, |
| "epoch": 47.275862068965516, |
| "grad_norm": 0.2348880465744704, |
| "kl": 0.33935546875, |
| "learning_rate": 1e-06, |
| "loss": 0.0078, |
| "step": 284 |
| }, |
| { |
| "batch_accuracy": 0.8348214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 236.69197845458984, |
| "epoch": 47.41379310344828, |
| "grad_norm": 0.6337068738976088, |
| "kl": 0.44580078125, |
| "learning_rate": 1e-06, |
| "loss": -0.0102, |
| "reward": 0.8348214626312256, |
| "reward_std": 0.05441322550177574, |
| "rewards/unified_reward_func": 0.8348214626312256, |
| "step": 285 |
| }, |
| { |
| "clip_ratio": 0.0008038270461838692, |
| "epoch": 47.55172413793103, |
| "grad_norm": 0.357288483372075, |
| "kl": 0.351806640625, |
| "learning_rate": 1e-06, |
| "loss": -0.0108, |
| "step": 286 |
| }, |
| { |
| "batch_accuracy": 0.9151785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 223.44644165039062, |
| "epoch": 47.689655172413794, |
| "grad_norm": 0.5834032629209837, |
| "kl": 0.589599609375, |
| "learning_rate": 1e-06, |
| "loss": -0.001, |
| "reward": 0.9151786118745804, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.9151786118745804, |
| "step": 287 |
| }, |
| { |
| "clip_ratio": 0.0004325698537286371, |
| "epoch": 47.827586206896555, |
| "grad_norm": 0.3715274484539023, |
| "kl": 0.354736328125, |
| "learning_rate": 1e-06, |
| "loss": -0.0017, |
| "step": 288 |
| }, |
| { |
| "batch_accuracy": 0.9107142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 247.15179061889648, |
| "epoch": 48.13793103448276, |
| "grad_norm": 0.8663466520277239, |
| "kl": 0.443359375, |
| "learning_rate": 1e-06, |
| "loss": 0.0129, |
| "reward": 0.9107143133878708, |
| "reward_std": 0.0417863167822361, |
| "rewards/unified_reward_func": 0.9107143133878708, |
| "step": 289 |
| }, |
| { |
| "clip_ratio": 0.0013505922979675233, |
| "epoch": 48.275862068965516, |
| "grad_norm": 0.47655967451195913, |
| "kl": 0.32861328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0122, |
| "step": 290 |
| }, |
| { |
| "batch_accuracy": 0.8035714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 243.96429443359375, |
| "epoch": 48.41379310344828, |
| "grad_norm": 0.6368598067476341, |
| "kl": 1.74658203125, |
| "learning_rate": 1e-06, |
| "loss": 0.0096, |
| "reward": 0.8035714626312256, |
| "reward_std": 0.03111080639064312, |
| "rewards/unified_reward_func": 0.8035714626312256, |
| "step": 291 |
| }, |
| { |
| "clip_ratio": 0.0009445616160519421, |
| "epoch": 48.55172413793103, |
| "grad_norm": 0.3813073979718291, |
| "kl": 0.836181640625, |
| "learning_rate": 1e-06, |
| "loss": 0.0089, |
| "step": 292 |
| }, |
| { |
| "batch_accuracy": 0.9017857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 216.31251525878906, |
| "epoch": 48.689655172413794, |
| "grad_norm": 0.565064019599756, |
| "kl": 0.4580078125, |
| "learning_rate": 1e-06, |
| "loss": -0.008, |
| "reward": 0.901785746216774, |
| "reward_std": 0.056364621967077255, |
| "rewards/unified_reward_func": 0.901785746216774, |
| "step": 293 |
| }, |
| { |
| "clip_ratio": 0.0011924125137738883, |
| "epoch": 48.827586206896555, |
| "grad_norm": 0.3460829859213167, |
| "kl": 0.314208984375, |
| "learning_rate": 1e-06, |
| "loss": -0.0086, |
| "step": 294 |
| }, |
| { |
| "batch_accuracy": 0.9196428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 217.0937614440918, |
| "epoch": 49.13793103448276, |
| "grad_norm": 0.27367858227193687, |
| "kl": 0.396484375, |
| "learning_rate": 1e-06, |
| "loss": -0.0041, |
| "reward": 0.9196428656578064, |
| "reward_std": 0.016532503068447113, |
| "rewards/unified_reward_func": 0.9196428656578064, |
| "step": 295 |
| }, |
| { |
| "clip_ratio": 0.0003673049795906991, |
| "epoch": 49.275862068965516, |
| "grad_norm": 0.1876339620606552, |
| "kl": 0.421630859375, |
| "learning_rate": 1e-06, |
| "loss": -0.0044, |
| "step": 296 |
| }, |
| { |
| "batch_accuracy": 0.875, |
| "clip_ratio": 0.0, |
| "completion_length": 230.0446548461914, |
| "epoch": 49.41379310344828, |
| "grad_norm": 0.40936720132784676, |
| "kl": 0.369873046875, |
| "learning_rate": 1e-06, |
| "loss": -0.0005, |
| "reward": 0.8750000298023224, |
| "reward_std": 0.0417863167822361, |
| "rewards/unified_reward_func": 0.8750000298023224, |
| "step": 297 |
| }, |
| { |
| "clip_ratio": 0.0008250945247709751, |
| "epoch": 49.55172413793103, |
| "grad_norm": 0.30886124746802324, |
| "kl": 0.391357421875, |
| "learning_rate": 1e-06, |
| "loss": -0.001, |
| "step": 298 |
| }, |
| { |
| "batch_accuracy": 0.8348214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 215.32590866088867, |
| "epoch": 49.689655172413794, |
| "grad_norm": 1.2565852570926377, |
| "kl": 0.72021484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0048, |
| "reward": 0.8348214626312256, |
| "reward_std": 0.04569191299378872, |
| "rewards/unified_reward_func": 0.8348214626312256, |
| "step": 299 |
| }, |
| { |
| "epoch": 49.827586206896555, |
| "grad_norm": 202.42169311010568, |
| "learning_rate": 1e-06, |
| "loss": 0.0688, |
| "step": 300 |
| }, |
| { |
| "epoch": 49.827586206896555, |
| "eval_batch_accuracy": 0.744047619047619, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 238.92111409505208, |
| "eval_kl": 0.09443359375, |
| "eval_loss": 0.007052370812743902, |
| "eval_reward": 0.7440476576487224, |
| "eval_reward_std": 0.1252529760201772, |
| "eval_rewards/unified_reward_func": 0.7440476576487224, |
| "eval_runtime": 295.3385, |
| "eval_samples_per_second": 0.339, |
| "eval_steps_per_second": 0.007, |
| "step": 300 |
| }, |
| { |
| "batch_accuracy": 0.8883928571428571, |
| "clip_ratio": 0.0004290657670935616, |
| "completion_length": 231.8526840209961, |
| "epoch": 50.13793103448276, |
| "grad_norm": 0.20282431054184938, |
| "kl": 0.21435546875, |
| "learning_rate": 1e-06, |
| "loss": -0.0047, |
| "reward": 0.8883928954601288, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.8883928954601288, |
| "step": 301 |
| }, |
| { |
| "clip_ratio": 5.189144212636165e-05, |
| "epoch": 50.275862068965516, |
| "grad_norm": 0.11943390673073986, |
| "kl": 0.081787109375, |
| "learning_rate": 1e-06, |
| "loss": -0.0049, |
| "step": 302 |
| }, |
| { |
| "batch_accuracy": 0.9196428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 231.25893783569336, |
| "epoch": 50.41379310344828, |
| "grad_norm": 0.3463004704729077, |
| "kl": 0.064453125, |
| "learning_rate": 1e-06, |
| "loss": -0.0109, |
| "reward": 0.9196428656578064, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9196428656578064, |
| "step": 303 |
| }, |
| { |
| "clip_ratio": 0.0006089565576985478, |
| "epoch": 50.55172413793103, |
| "grad_norm": 0.17425565901723739, |
| "kl": 0.06292724609375, |
| "learning_rate": 1e-06, |
| "loss": -0.0114, |
| "step": 304 |
| }, |
| { |
| "batch_accuracy": 0.7767857142857142, |
| "clip_ratio": 0.0, |
| "completion_length": 231.24108505249023, |
| "epoch": 50.689655172413794, |
| "grad_norm": 0.43862146523548484, |
| "kl": 0.1053466796875, |
| "learning_rate": 1e-06, |
| "loss": 0.005, |
| "reward": 0.7767857611179352, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.7767857611179352, |
| "step": 305 |
| }, |
| { |
| "clip_ratio": 0.0006555429135914892, |
| "epoch": 50.827586206896555, |
| "grad_norm": 0.24700514710664973, |
| "kl": 0.1103515625, |
| "learning_rate": 1e-06, |
| "loss": 0.0045, |
| "step": 306 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 233.2321548461914, |
| "epoch": 51.13793103448276, |
| "grad_norm": 0.39954111913085166, |
| "kl": 0.0909423828125, |
| "learning_rate": 1e-06, |
| "loss": -0.0034, |
| "reward": 0.8482143133878708, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8482143133878708, |
| "step": 307 |
| }, |
| { |
| "clip_ratio": 0.000625371903879568, |
| "epoch": 51.275862068965516, |
| "grad_norm": 0.17080867054188612, |
| "kl": 0.0911865234375, |
| "learning_rate": 1e-06, |
| "loss": -0.004, |
| "step": 308 |
| }, |
| { |
| "batch_accuracy": 0.7946428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 280.04019927978516, |
| "epoch": 51.41379310344828, |
| "grad_norm": 0.47116099169757397, |
| "kl": 0.197998046875, |
| "learning_rate": 1e-06, |
| "loss": -0.0044, |
| "reward": 0.7946428954601288, |
| "reward_std": 0.05831881985068321, |
| "rewards/unified_reward_func": 0.7946428954601288, |
| "step": 309 |
| }, |
| { |
| "clip_ratio": 0.0012788574094884098, |
| "epoch": 51.55172413793103, |
| "grad_norm": 0.30029225829463274, |
| "kl": 0.19482421875, |
| "learning_rate": 1e-06, |
| "loss": -0.005, |
| "step": 310 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 186.85715103149414, |
| "epoch": 51.689655172413794, |
| "grad_norm": 0.4418548735106828, |
| "kl": 0.100341796875, |
| "learning_rate": 1e-06, |
| "loss": 0.0075, |
| "reward": 0.8482143133878708, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8482143133878708, |
| "step": 311 |
| }, |
| { |
| "clip_ratio": 0.00036868505412712693, |
| "epoch": 51.827586206896555, |
| "grad_norm": 0.2913408678197493, |
| "kl": 0.1114501953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0069, |
| "step": 312 |
| }, |
| { |
| "batch_accuracy": 0.8303571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 280.5044746398926, |
| "epoch": 52.13793103448276, |
| "grad_norm": 1.4439378816929274, |
| "kl": 0.15966796875, |
| "learning_rate": 1e-06, |
| "loss": 0.0152, |
| "reward": 0.8303571790456772, |
| "reward_std": 0.09229394607245922, |
| "rewards/unified_reward_func": 0.8303571790456772, |
| "step": 313 |
| }, |
| { |
| "clip_ratio": 0.00311722481274046, |
| "epoch": 52.275862068965516, |
| "grad_norm": 0.8123430006245371, |
| "kl": 0.193603515625, |
| "learning_rate": 1e-06, |
| "loss": 0.0141, |
| "step": 314 |
| }, |
| { |
| "batch_accuracy": 0.8571428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 222.40625381469727, |
| "epoch": 52.41379310344828, |
| "grad_norm": 0.04415438982672496, |
| "kl": 0.1348876953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0001, |
| "reward": 0.8571428954601288, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.8571428954601288, |
| "step": 315 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 52.55172413793103, |
| "grad_norm": 0.04462437267095938, |
| "kl": 0.135498046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0001, |
| "step": 316 |
| }, |
| { |
| "batch_accuracy": 0.9107142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 205.35715103149414, |
| "epoch": 52.689655172413794, |
| "grad_norm": 0.5442733579616569, |
| "kl": 0.173828125, |
| "learning_rate": 1e-06, |
| "loss": 0.0143, |
| "reward": 0.910714328289032, |
| "reward_std": 0.0417863167822361, |
| "rewards/unified_reward_func": 0.910714328289032, |
| "step": 317 |
| }, |
| { |
| "clip_ratio": 0.0013521471119020134, |
| "epoch": 52.827586206896555, |
| "grad_norm": 0.47979492418943703, |
| "kl": 0.226318359375, |
| "learning_rate": 1e-06, |
| "loss": 0.0136, |
| "step": 318 |
| }, |
| { |
| "batch_accuracy": 0.84375, |
| "clip_ratio": 0.0, |
| "completion_length": 231.1116180419922, |
| "epoch": 53.13793103448276, |
| "grad_norm": 0.9586342201910061, |
| "kl": 0.2431640625, |
| "learning_rate": 1e-06, |
| "loss": 0.0024, |
| "reward": 0.8437500447034836, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.8437500447034836, |
| "step": 319 |
| }, |
| { |
| "clip_ratio": 0.001583009579917416, |
| "epoch": 53.275862068965516, |
| "grad_norm": 0.4632957681239461, |
| "kl": 0.375244140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0017, |
| "step": 320 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 242.8750114440918, |
| "epoch": 53.41379310344828, |
| "grad_norm": 1.084233760954395, |
| "kl": 0.63623046875, |
| "learning_rate": 1e-06, |
| "loss": -0.0007, |
| "reward": 0.8839286267757416, |
| "reward_std": 0.06704013235867023, |
| "rewards/unified_reward_func": 0.8839286267757416, |
| "step": 321 |
| }, |
| { |
| "clip_ratio": 0.002837361069396138, |
| "epoch": 53.55172413793103, |
| "grad_norm": 0.8171731868686768, |
| "kl": 0.8486328125, |
| "learning_rate": 1e-06, |
| "loss": -0.0015, |
| "step": 322 |
| }, |
| { |
| "batch_accuracy": 0.8169642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 245.22322463989258, |
| "epoch": 53.689655172413794, |
| "grad_norm": 0.4003946931627436, |
| "kl": 0.43408203125, |
| "learning_rate": 1e-06, |
| "loss": -0.0023, |
| "reward": 0.816964328289032, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.816964328289032, |
| "step": 323 |
| }, |
| { |
| "clip_ratio": 0.00021813277271576226, |
| "epoch": 53.827586206896555, |
| "grad_norm": 0.1998574216227568, |
| "kl": 0.41748046875, |
| "learning_rate": 1e-06, |
| "loss": -0.0026, |
| "step": 324 |
| }, |
| { |
| "batch_accuracy": 0.8660714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 305.17858123779297, |
| "epoch": 54.13793103448276, |
| "grad_norm": 0.6691039836172034, |
| "kl": 0.56787109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0028, |
| "reward": 0.8660714626312256, |
| "reward_std": 0.056364621967077255, |
| "rewards/unified_reward_func": 0.8660714626312256, |
| "step": 325 |
| }, |
| { |
| "clip_ratio": 0.0009606677340343595, |
| "epoch": 54.275862068965516, |
| "grad_norm": 0.744355757067827, |
| "kl": 0.39453125, |
| "learning_rate": 1e-06, |
| "loss": 0.0023, |
| "step": 326 |
| }, |
| { |
| "batch_accuracy": 0.8080357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 227.95983123779297, |
| "epoch": 54.41379310344828, |
| "grad_norm": 0.6353132669066994, |
| "kl": 0.395751953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0037, |
| "reward": 0.8080357313156128, |
| "reward_std": 0.04373771324753761, |
| "rewards/unified_reward_func": 0.8080357313156128, |
| "step": 327 |
| }, |
| { |
| "clip_ratio": 0.00077925888763275, |
| "epoch": 54.55172413793103, |
| "grad_norm": 0.5180919767758974, |
| "kl": 0.447509765625, |
| "learning_rate": 1e-06, |
| "loss": 0.003, |
| "step": 328 |
| }, |
| { |
| "batch_accuracy": 0.9196428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 226.0000114440918, |
| "epoch": 54.689655172413794, |
| "grad_norm": 0.40607137186611947, |
| "kl": 0.305419921875, |
| "learning_rate": 1e-06, |
| "loss": 0.0036, |
| "reward": 0.9196428656578064, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9196428656578064, |
| "step": 329 |
| }, |
| { |
| "clip_ratio": 0.0003793864598264918, |
| "epoch": 54.827586206896555, |
| "grad_norm": 0.29914570877843377, |
| "kl": 0.2880859375, |
| "learning_rate": 1e-06, |
| "loss": 0.003, |
| "step": 330 |
| }, |
| { |
| "batch_accuracy": 0.7321428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 280.06697845458984, |
| "epoch": 55.13793103448276, |
| "grad_norm": 0.8891067679376151, |
| "kl": 0.72314453125, |
| "learning_rate": 1e-06, |
| "loss": -0.0131, |
| "reward": 0.7321428805589676, |
| "reward_std": 0.05050762556493282, |
| "rewards/unified_reward_func": 0.7321428805589676, |
| "step": 331 |
| }, |
| { |
| "clip_ratio": 0.0006313109188340604, |
| "epoch": 55.275862068965516, |
| "grad_norm": 0.4728098972900883, |
| "kl": 0.37744140625, |
| "learning_rate": 1e-06, |
| "loss": -0.014, |
| "step": 332 |
| }, |
| { |
| "batch_accuracy": 0.8571428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 207.90625762939453, |
| "epoch": 55.41379310344828, |
| "grad_norm": 0.6588124033082572, |
| "kl": 0.29248046875, |
| "learning_rate": 1e-06, |
| "loss": -0.0149, |
| "reward": 0.8571428805589676, |
| "reward_std": 0.050200898200273514, |
| "rewards/unified_reward_func": 0.8571428805589676, |
| "step": 333 |
| }, |
| { |
| "clip_ratio": 0.0005623959586955607, |
| "epoch": 55.55172413793103, |
| "grad_norm": 0.9901216280739612, |
| "kl": 0.21533203125, |
| "learning_rate": 1e-06, |
| "loss": -0.0155, |
| "step": 334 |
| }, |
| { |
| "batch_accuracy": 0.8928571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 235.65179443359375, |
| "epoch": 55.689655172413794, |
| "grad_norm": 0.8589161497188839, |
| "kl": 0.422607421875, |
| "learning_rate": 1e-06, |
| "loss": 0.0113, |
| "reward": 0.8928571790456772, |
| "reward_std": 0.08161843568086624, |
| "rewards/unified_reward_func": 0.8928571790456772, |
| "step": 335 |
| }, |
| { |
| "clip_ratio": 0.0015647939726477489, |
| "epoch": 55.827586206896555, |
| "grad_norm": 286.42521108884824, |
| "kl": 0.23583984375, |
| "learning_rate": 1e-06, |
| "loss": 0.0848, |
| "step": 336 |
| }, |
| { |
| "batch_accuracy": 0.9910714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 221.1785774230957, |
| "epoch": 56.13793103448276, |
| "grad_norm": 0.47598821491528226, |
| "kl": 0.222900390625, |
| "learning_rate": 1e-06, |
| "loss": -0.0022, |
| "reward": 0.9910714626312256, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9910714626312256, |
| "step": 337 |
| }, |
| { |
| "clip_ratio": 0.0003239207435399294, |
| "epoch": 56.275862068965516, |
| "grad_norm": 0.26568970245571083, |
| "kl": 0.291259765625, |
| "learning_rate": 1e-06, |
| "loss": -0.0027, |
| "step": 338 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 230.15625762939453, |
| "epoch": 56.41379310344828, |
| "grad_norm": 4.759978452673977, |
| "kl": 1.978271484375, |
| "learning_rate": 1e-06, |
| "loss": -0.0017, |
| "reward": 0.8839286267757416, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839286267757416, |
| "step": 339 |
| }, |
| { |
| "clip_ratio": 0.00047356344293802977, |
| "epoch": 56.55172413793103, |
| "grad_norm": 0.48573771123545095, |
| "kl": 0.251220703125, |
| "learning_rate": 1e-06, |
| "loss": -0.0034, |
| "step": 340 |
| }, |
| { |
| "batch_accuracy": 0.6785714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 284.6651916503906, |
| "epoch": 56.689655172413794, |
| "grad_norm": 0.39208663367499275, |
| "kl": 0.23291015625, |
| "learning_rate": 1e-06, |
| "loss": -0.0069, |
| "reward": 0.6785714626312256, |
| "reward_std": 0.03696779906749725, |
| "rewards/unified_reward_func": 0.6785714626312256, |
| "step": 341 |
| }, |
| { |
| "clip_ratio": 0.00026187896582996473, |
| "epoch": 56.827586206896555, |
| "grad_norm": 0.312513512345737, |
| "kl": 0.2509765625, |
| "learning_rate": 1e-06, |
| "loss": -0.0076, |
| "step": 342 |
| }, |
| { |
| "batch_accuracy": 0.8883928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 243.75447845458984, |
| "epoch": 57.13793103448276, |
| "grad_norm": 24.52458761423547, |
| "kl": 6.576416015625, |
| "learning_rate": 1e-06, |
| "loss": 0.0041, |
| "reward": 0.8883928805589676, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.8883928805589676, |
| "step": 343 |
| }, |
| { |
| "clip_ratio": 0.0012561257171910256, |
| "epoch": 57.275862068965516, |
| "grad_norm": 0.534774035654984, |
| "kl": 0.287109375, |
| "learning_rate": 1e-06, |
| "loss": -0.0022, |
| "step": 344 |
| }, |
| { |
| "batch_accuracy": 0.8169642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 262.10268783569336, |
| "epoch": 57.41379310344828, |
| "grad_norm": 4.2828878662326595, |
| "kl": 2.083984375, |
| "learning_rate": 1e-06, |
| "loss": 0.0042, |
| "reward": 0.8169643431901932, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.8169643431901932, |
| "step": 345 |
| }, |
| { |
| "clip_ratio": 0.00011598970741033554, |
| "epoch": 57.55172413793103, |
| "grad_norm": 0.25471420002458783, |
| "kl": 0.4111328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0026, |
| "step": 346 |
| }, |
| { |
| "batch_accuracy": 0.9241071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 216.49554061889648, |
| "epoch": 57.689655172413794, |
| "grad_norm": 0.40655736639617174, |
| "kl": 0.369873046875, |
| "learning_rate": 1e-06, |
| "loss": -0.0014, |
| "reward": 0.9241071939468384, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.9241071939468384, |
| "step": 347 |
| }, |
| { |
| "clip_ratio": 9.582215716363862e-05, |
| "epoch": 57.827586206896555, |
| "grad_norm": 0.20228576025102957, |
| "kl": 0.185302734375, |
| "learning_rate": 1e-06, |
| "loss": -0.0019, |
| "step": 348 |
| }, |
| { |
| "batch_accuracy": 0.9508928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 223.70983123779297, |
| "epoch": 58.13793103448276, |
| "grad_norm": 0.3934745552241851, |
| "kl": 0.159912109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0105, |
| "reward": 0.9508928954601288, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.9508928954601288, |
| "step": 349 |
| }, |
| { |
| "epoch": 58.275862068965516, |
| "grad_norm": 0.47304722160177515, |
| "learning_rate": 1e-06, |
| "loss": 0.01, |
| "step": 350 |
| }, |
| { |
| "epoch": 58.275862068965516, |
| "eval_batch_accuracy": 0.7678571428571428, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 250.05552876790364, |
| "eval_kl": 0.5563151041666666, |
| "eval_loss": -0.0007301162695512176, |
| "eval_reward": 0.7678571780522664, |
| "eval_reward_std": 0.11819532414277395, |
| "eval_rewards/unified_reward_func": 0.7678571780522664, |
| "eval_runtime": 299.8452, |
| "eval_samples_per_second": 0.334, |
| "eval_steps_per_second": 0.007, |
| "step": 350 |
| }, |
| { |
| "batch_accuracy": 0.7946428571428571, |
| "clip_ratio": 0.0006118576420703903, |
| "completion_length": 244.39733123779297, |
| "epoch": 58.41379310344828, |
| "grad_norm": 0.3158750121263176, |
| "kl": 0.20098876953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0033, |
| "reward": 0.7946428954601288, |
| "reward_std": 0.016532503068447113, |
| "rewards/unified_reward_func": 0.7946428954601288, |
| "step": 351 |
| }, |
| { |
| "clip_ratio": 0.00023174374655354768, |
| "epoch": 58.55172413793103, |
| "grad_norm": 0.2202251953260023, |
| "kl": 0.1893310546875, |
| "learning_rate": 1e-06, |
| "loss": 0.0029, |
| "step": 352 |
| }, |
| { |
| "batch_accuracy": 0.7946428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 243.59822463989258, |
| "epoch": 58.689655172413794, |
| "grad_norm": 0.6233607920605455, |
| "kl": 0.64697265625, |
| "learning_rate": 1e-06, |
| "loss": -0.0133, |
| "reward": 0.7946428954601288, |
| "reward_std": 0.05831881985068321, |
| "rewards/unified_reward_func": 0.7946428954601288, |
| "step": 353 |
| }, |
| { |
| "clip_ratio": 0.0008263020572485402, |
| "epoch": 58.827586206896555, |
| "grad_norm": 0.4362430398528479, |
| "kl": 0.450439453125, |
| "learning_rate": 1e-06, |
| "loss": -0.0139, |
| "step": 354 |
| }, |
| { |
| "batch_accuracy": 0.8794642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 228.5357208251953, |
| "epoch": 59.13793103448276, |
| "grad_norm": 0.37929784934812716, |
| "kl": 0.2900390625, |
| "learning_rate": 1e-06, |
| "loss": -0.0112, |
| "reward": 0.879464328289032, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.879464328289032, |
| "step": 355 |
| }, |
| { |
| "clip_ratio": 0.0007167806033976376, |
| "epoch": 59.275862068965516, |
| "grad_norm": 3.93509932542601, |
| "kl": 0.1904296875, |
| "learning_rate": 1e-06, |
| "loss": -0.0104, |
| "step": 356 |
| }, |
| { |
| "batch_accuracy": 0.9464285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 204.1250114440918, |
| "epoch": 59.41379310344828, |
| "grad_norm": 0.38429841165810424, |
| "kl": 0.166748046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0011, |
| "reward": 0.9464285969734192, |
| "reward_std": 0.033065006136894226, |
| "rewards/unified_reward_func": 0.9464285969734192, |
| "step": 357 |
| }, |
| { |
| "clip_ratio": 0.0006046723137842491, |
| "epoch": 59.55172413793103, |
| "grad_norm": 0.33219320585174655, |
| "kl": 0.1893310546875, |
| "learning_rate": 1e-06, |
| "loss": 0.0007, |
| "step": 358 |
| }, |
| { |
| "batch_accuracy": 0.7633928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 262.7410888671875, |
| "epoch": 59.689655172413794, |
| "grad_norm": 0.6733029338519297, |
| "kl": 0.2568359375, |
| "learning_rate": 1e-06, |
| "loss": -0.015, |
| "reward": 0.7633928954601288, |
| "reward_std": 0.05441322363913059, |
| "rewards/unified_reward_func": 0.7633928954601288, |
| "step": 359 |
| }, |
| { |
| "clip_ratio": 0.0009442013979423791, |
| "epoch": 59.827586206896555, |
| "grad_norm": 0.5664337727116652, |
| "kl": 0.290283203125, |
| "learning_rate": 1e-06, |
| "loss": -0.0154, |
| "step": 360 |
| }, |
| { |
| "batch_accuracy": 0.9107142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 209.6651840209961, |
| "epoch": 60.13793103448276, |
| "grad_norm": 0.6726035076697437, |
| "kl": 0.287841796875, |
| "learning_rate": 1e-06, |
| "loss": -0.0003, |
| "reward": 0.910714328289032, |
| "reward_std": 0.0417863167822361, |
| "rewards/unified_reward_func": 0.910714328289032, |
| "step": 361 |
| }, |
| { |
| "clip_ratio": 0.0008161436417140067, |
| "epoch": 60.275862068965516, |
| "grad_norm": 0.4014200013710924, |
| "kl": 0.245849609375, |
| "learning_rate": 1e-06, |
| "loss": -0.0013, |
| "step": 362 |
| }, |
| { |
| "batch_accuracy": 0.9196428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 257.308048248291, |
| "epoch": 60.41379310344828, |
| "grad_norm": 0.5910338904683496, |
| "kl": 0.564208984375, |
| "learning_rate": 1e-06, |
| "loss": -0.0033, |
| "reward": 0.9196428954601288, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9196428954601288, |
| "step": 363 |
| }, |
| { |
| "clip_ratio": 0.0003132874990114942, |
| "epoch": 60.55172413793103, |
| "grad_norm": 0.1909713517011228, |
| "kl": 0.2470703125, |
| "learning_rate": 1e-06, |
| "loss": -0.0037, |
| "step": 364 |
| }, |
| { |
| "batch_accuracy": 0.7723214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 232.00447463989258, |
| "epoch": 60.689655172413794, |
| "grad_norm": 0.43041414128294087, |
| "kl": 0.199462890625, |
| "learning_rate": 1e-06, |
| "loss": 0.0035, |
| "reward": 0.7723214477300644, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.7723214477300644, |
| "step": 365 |
| }, |
| { |
| "clip_ratio": 0.0005480971594806761, |
| "epoch": 60.827586206896555, |
| "grad_norm": 0.2982648433527683, |
| "kl": 0.2138671875, |
| "learning_rate": 1e-06, |
| "loss": 0.0029, |
| "step": 366 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 193.9598274230957, |
| "epoch": 61.13793103448276, |
| "grad_norm": 143.25192413352647, |
| "kl": 55.736328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0511, |
| "reward": 0.8839285969734192, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839285969734192, |
| "step": 367 |
| }, |
| { |
| "clip_ratio": 0.0007856006559450179, |
| "epoch": 61.275862068965516, |
| "grad_norm": 215.10922197761653, |
| "kl": 0.515625, |
| "learning_rate": 1e-06, |
| "loss": 0.1139, |
| "step": 368 |
| }, |
| { |
| "batch_accuracy": 0.8660714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 251.83929443359375, |
| "epoch": 61.41379310344828, |
| "grad_norm": 0.749264934206035, |
| "kl": 0.57666015625, |
| "learning_rate": 1e-06, |
| "loss": 0.0052, |
| "reward": 0.8660714775323868, |
| "reward_std": 0.05831881985068321, |
| "rewards/unified_reward_func": 0.8660714775323868, |
| "step": 369 |
| }, |
| { |
| "clip_ratio": 0.0008260020840680227, |
| "epoch": 61.55172413793103, |
| "grad_norm": 0.4966394103134508, |
| "kl": 0.284912109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0041, |
| "step": 370 |
| }, |
| { |
| "batch_accuracy": 0.9241071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 231.1250114440918, |
| "epoch": 61.689655172413794, |
| "grad_norm": 0.43740040556746523, |
| "kl": 0.4306640625, |
| "learning_rate": 1e-06, |
| "loss": -0.0017, |
| "reward": 0.924107164144516, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.924107164144516, |
| "step": 371 |
| }, |
| { |
| "clip_ratio": 7.185972935985774e-05, |
| "epoch": 61.827586206896555, |
| "grad_norm": 0.23887790320111377, |
| "kl": 0.353515625, |
| "learning_rate": 1e-06, |
| "loss": -0.002, |
| "step": 372 |
| }, |
| { |
| "batch_accuracy": 0.7723214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 247.37054443359375, |
| "epoch": 62.13793103448276, |
| "grad_norm": 0.6206085533572123, |
| "kl": 0.36083984375, |
| "learning_rate": 1e-06, |
| "loss": 0.0151, |
| "reward": 0.7723214775323868, |
| "reward_std": 0.03788071870803833, |
| "rewards/unified_reward_func": 0.7723214775323868, |
| "step": 373 |
| }, |
| { |
| "clip_ratio": 0.001170877949334681, |
| "epoch": 62.275862068965516, |
| "grad_norm": 0.34783589826385763, |
| "kl": 0.228515625, |
| "learning_rate": 1e-06, |
| "loss": 0.0145, |
| "step": 374 |
| }, |
| { |
| "batch_accuracy": 0.8928571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 226.1250114440918, |
| "epoch": 62.41379310344828, |
| "grad_norm": 0.33943189999115525, |
| "kl": 0.429931640625, |
| "learning_rate": 1e-06, |
| "loss": 0.0004, |
| "reward": 0.8928571939468384, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.8928571939468384, |
| "step": 375 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 62.55172413793103, |
| "grad_norm": 0.04327261789576398, |
| "kl": 0.22802734375, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 376 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 240.47322845458984, |
| "epoch": 62.689655172413794, |
| "grad_norm": 0.506391391627102, |
| "kl": 0.338623046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0074, |
| "reward": 0.8839285969734192, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839285969734192, |
| "step": 377 |
| }, |
| { |
| "clip_ratio": 0.0006104611675255001, |
| "epoch": 62.827586206896555, |
| "grad_norm": 0.2965824495255123, |
| "kl": 0.333251953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0065, |
| "step": 378 |
| }, |
| { |
| "batch_accuracy": 0.9642857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 226.99108123779297, |
| "epoch": 63.13793103448276, |
| "grad_norm": 0.06203814627270833, |
| "kl": 0.3408203125, |
| "learning_rate": 1e-06, |
| "loss": 0.0003, |
| "reward": 0.9642857313156128, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.9642857313156128, |
| "step": 379 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 63.275862068965516, |
| "grad_norm": 0.045239787224193535, |
| "kl": 0.284912109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0003, |
| "step": 380 |
| }, |
| { |
| "batch_accuracy": 0.7991071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 235.86608123779297, |
| "epoch": 63.41379310344828, |
| "grad_norm": 0.3996258785535166, |
| "kl": 0.344970703125, |
| "learning_rate": 1e-06, |
| "loss": -0.0054, |
| "reward": 0.7991071790456772, |
| "reward_std": 0.03501640260219574, |
| "rewards/unified_reward_func": 0.7991071790456772, |
| "step": 381 |
| }, |
| { |
| "clip_ratio": 0.0003395631938474253, |
| "epoch": 63.55172413793103, |
| "grad_norm": 0.3568904247413468, |
| "kl": 0.316162109375, |
| "learning_rate": 1e-06, |
| "loss": -0.006, |
| "step": 382 |
| }, |
| { |
| "batch_accuracy": 0.8616071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 212.32590103149414, |
| "epoch": 63.689655172413794, |
| "grad_norm": 0.6000615523479406, |
| "kl": 0.24365234375, |
| "learning_rate": 1e-06, |
| "loss": -0.0073, |
| "reward": 0.8616071790456772, |
| "reward_std": 0.060270218178629875, |
| "rewards/unified_reward_func": 0.8616071790456772, |
| "step": 383 |
| }, |
| { |
| "clip_ratio": 0.0009390746981807752, |
| "epoch": 63.827586206896555, |
| "grad_norm": 0.3933344938486858, |
| "kl": 0.2490234375, |
| "learning_rate": 1e-06, |
| "loss": -0.0082, |
| "step": 384 |
| }, |
| { |
| "batch_accuracy": 0.90625, |
| "clip_ratio": 0.0, |
| "completion_length": 250.05358505249023, |
| "epoch": 64.13793103448276, |
| "grad_norm": 0.8330463899634639, |
| "kl": 0.247802734375, |
| "learning_rate": 1e-06, |
| "loss": 0.01, |
| "reward": 0.9062500298023224, |
| "reward_std": 0.05441322550177574, |
| "rewards/unified_reward_func": 0.9062500298023224, |
| "step": 385 |
| }, |
| { |
| "clip_ratio": 0.0013701908756047487, |
| "epoch": 64.27586206896552, |
| "grad_norm": 0.47745481228963144, |
| "kl": 0.24755859375, |
| "learning_rate": 1e-06, |
| "loss": 0.009, |
| "step": 386 |
| }, |
| { |
| "batch_accuracy": 0.84375, |
| "clip_ratio": 0.0, |
| "completion_length": 218.30804824829102, |
| "epoch": 64.41379310344827, |
| "grad_norm": 0.4860256067494172, |
| "kl": 0.374267578125, |
| "learning_rate": 1e-06, |
| "loss": -0.0047, |
| "reward": 0.8437500298023224, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.8437500298023224, |
| "step": 387 |
| }, |
| { |
| "clip_ratio": 0.0006055501580704004, |
| "epoch": 64.55172413793103, |
| "grad_norm": 0.2341605321475626, |
| "kl": 0.273193359375, |
| "learning_rate": 1e-06, |
| "loss": -0.0051, |
| "step": 388 |
| }, |
| { |
| "batch_accuracy": 0.875, |
| "clip_ratio": 0.0, |
| "completion_length": 222.17858123779297, |
| "epoch": 64.6896551724138, |
| "grad_norm": 0.5574211120856241, |
| "kl": 0.27490234375, |
| "learning_rate": 1e-06, |
| "loss": -0.0048, |
| "reward": 0.8750000298023224, |
| "reward_std": 0.05050762742757797, |
| "rewards/unified_reward_func": 0.8750000298023224, |
| "step": 389 |
| }, |
| { |
| "clip_ratio": 0.0009339429816463962, |
| "epoch": 64.82758620689656, |
| "grad_norm": 0.514534910574207, |
| "kl": 0.236083984375, |
| "learning_rate": 1e-06, |
| "loss": -0.0053, |
| "step": 390 |
| }, |
| { |
| "batch_accuracy": 0.8392857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 228.56697845458984, |
| "epoch": 65.13793103448276, |
| "grad_norm": 0.5745364965962138, |
| "kl": 0.2763671875, |
| "learning_rate": 1e-06, |
| "loss": -0.0238, |
| "reward": 0.839285746216774, |
| "reward_std": 0.05050762742757797, |
| "rewards/unified_reward_func": 0.839285746216774, |
| "step": 391 |
| }, |
| { |
| "clip_ratio": 0.0009972895204555243, |
| "epoch": 65.27586206896552, |
| "grad_norm": 0.33961545508933666, |
| "kl": 0.28369140625, |
| "learning_rate": 1e-06, |
| "loss": -0.0245, |
| "step": 392 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 223.16518783569336, |
| "epoch": 65.41379310344827, |
| "grad_norm": 0.3434587742819441, |
| "kl": 0.2099609375, |
| "learning_rate": 1e-06, |
| "loss": -0.0005, |
| "reward": 0.8482143431901932, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8482143431901932, |
| "step": 393 |
| }, |
| { |
| "clip_ratio": 0.0007103000534698367, |
| "epoch": 65.55172413793103, |
| "grad_norm": 0.20376672707462493, |
| "kl": 0.248046875, |
| "learning_rate": 1e-06, |
| "loss": -0.0007, |
| "step": 394 |
| }, |
| { |
| "batch_accuracy": 0.9196428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 214.17858123779297, |
| "epoch": 65.6896551724138, |
| "grad_norm": 0.42867069975063926, |
| "kl": 0.181396484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0072, |
| "reward": 0.9196428954601288, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9196428954601288, |
| "step": 395 |
| }, |
| { |
| "clip_ratio": 0.00042697125172708184, |
| "epoch": 65.82758620689656, |
| "grad_norm": 0.20848546236555793, |
| "kl": 0.192626953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0067, |
| "step": 396 |
| }, |
| { |
| "batch_accuracy": 0.84375, |
| "clip_ratio": 0.0, |
| "completion_length": 208.79019165039062, |
| "epoch": 66.13793103448276, |
| "grad_norm": 0.5565175606934627, |
| "kl": 0.481201171875, |
| "learning_rate": 1e-06, |
| "loss": 0.0012, |
| "reward": 0.8437500298023224, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.8437500298023224, |
| "step": 397 |
| }, |
| { |
| "clip_ratio": 0.0008260089380200952, |
| "epoch": 66.27586206896552, |
| "grad_norm": 0.3524931608816777, |
| "kl": 0.369384765625, |
| "learning_rate": 1e-06, |
| "loss": 0.0004, |
| "step": 398 |
| }, |
| { |
| "batch_accuracy": 0.8660714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 250.0178680419922, |
| "epoch": 66.41379310344827, |
| "grad_norm": 0.4351526947834308, |
| "kl": 0.41015625, |
| "learning_rate": 1e-06, |
| "loss": -0.0251, |
| "reward": 0.8660714477300644, |
| "reward_std": 0.04434390366077423, |
| "rewards/unified_reward_func": 0.8660714477300644, |
| "step": 399 |
| }, |
| { |
| "epoch": 66.55172413793103, |
| "grad_norm": 0.24679700643125732, |
| "learning_rate": 1e-06, |
| "loss": -0.0257, |
| "step": 400 |
| }, |
| { |
| "epoch": 66.55172413793103, |
| "eval_batch_accuracy": 0.7511904761904762, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 248.59719746907552, |
| "eval_kl": 0.07989908854166666, |
| "eval_loss": 0.016887282952666283, |
| "eval_reward": 0.7511905193328857, |
| "eval_reward_std": 0.12925923814376195, |
| "eval_rewards/unified_reward_func": 0.7511905193328857, |
| "eval_runtime": 296.0887, |
| "eval_samples_per_second": 0.338, |
| "eval_steps_per_second": 0.007, |
| "step": 400 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.00014136406025500037, |
| "completion_length": 253.33037185668945, |
| "epoch": 66.6896551724138, |
| "grad_norm": 0.31349476916552094, |
| "kl": 0.20245361328125, |
| "learning_rate": 1e-06, |
| "loss": -0.0008, |
| "reward": 0.8839286267757416, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839286267757416, |
| "step": 401 |
| }, |
| { |
| "clip_ratio": 0.00036735970206791535, |
| "epoch": 66.82758620689656, |
| "grad_norm": 0.19451669827626714, |
| "kl": 0.0814208984375, |
| "learning_rate": 1e-06, |
| "loss": -0.0014, |
| "step": 402 |
| }, |
| { |
| "batch_accuracy": 0.8883928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 215.8839340209961, |
| "epoch": 67.13793103448276, |
| "grad_norm": 0.38417199755616704, |
| "kl": 0.0771484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0008, |
| "reward": 0.8883928954601288, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.8883928954601288, |
| "step": 403 |
| }, |
| { |
| "clip_ratio": 8.53727906360291e-05, |
| "epoch": 67.27586206896552, |
| "grad_norm": 0.18756072191965203, |
| "kl": 0.081298828125, |
| "learning_rate": 1e-06, |
| "loss": 0.0003, |
| "step": 404 |
| }, |
| { |
| "batch_accuracy": 0.9598214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 225.5178680419922, |
| "epoch": 67.41379310344827, |
| "grad_norm": 0.3038476247906523, |
| "kl": 0.1053466796875, |
| "learning_rate": 1e-06, |
| "loss": 0.0025, |
| "reward": 0.9598214626312256, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.9598214626312256, |
| "step": 405 |
| }, |
| { |
| "clip_ratio": 0.00034553295699879527, |
| "epoch": 67.55172413793103, |
| "grad_norm": 0.17955543384066075, |
| "kl": 0.1016845703125, |
| "learning_rate": 1e-06, |
| "loss": 0.0021, |
| "step": 406 |
| }, |
| { |
| "batch_accuracy": 0.8303571428571429, |
| "clip_ratio": 0.0, |
| "completion_length": 266.62500381469727, |
| "epoch": 67.6896551724138, |
| "grad_norm": 0.40599740725102873, |
| "kl": 0.1082763671875, |
| "learning_rate": 1e-06, |
| "loss": -0.0176, |
| "reward": 0.830357164144516, |
| "reward_std": 0.056364621967077255, |
| "rewards/unified_reward_func": 0.830357164144516, |
| "step": 407 |
| }, |
| { |
| "clip_ratio": 0.0004739694340969436, |
| "epoch": 67.82758620689656, |
| "grad_norm": 0.32870338345791006, |
| "kl": 0.10009765625, |
| "learning_rate": 1e-06, |
| "loss": -0.0182, |
| "step": 408 |
| }, |
| { |
| "batch_accuracy": 0.9151785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 243.8616180419922, |
| "epoch": 68.13793103448276, |
| "grad_norm": 0.516708055153792, |
| "kl": 0.0828857421875, |
| "learning_rate": 1e-06, |
| "loss": 0.0014, |
| "reward": 0.9151785969734192, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.9151785969734192, |
| "step": 409 |
| }, |
| { |
| "clip_ratio": 0.00027526616759132594, |
| "epoch": 68.27586206896552, |
| "grad_norm": 0.28028763512050475, |
| "kl": 0.0921630859375, |
| "learning_rate": 1e-06, |
| "loss": 0.0007, |
| "step": 410 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 245.70983123779297, |
| "epoch": 68.41379310344827, |
| "grad_norm": 0.47997056233420055, |
| "kl": 0.0889892578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0009, |
| "reward": 0.848214328289032, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.848214328289032, |
| "step": 411 |
| }, |
| { |
| "clip_ratio": 0.0004889596821158193, |
| "epoch": 68.55172413793103, |
| "grad_norm": 0.23762324351012565, |
| "kl": 0.1072998046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0004, |
| "step": 412 |
| }, |
| { |
| "batch_accuracy": 0.9241071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 208.18304443359375, |
| "epoch": 68.6896551724138, |
| "grad_norm": 0.7607902677997762, |
| "kl": 0.1046142578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0113, |
| "reward": 0.9241071939468384, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.9241071939468384, |
| "step": 413 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 68.82758620689656, |
| "grad_norm": 0.9318297670937001, |
| "kl": 0.112548828125, |
| "learning_rate": 1e-06, |
| "loss": 0.0106, |
| "step": 414 |
| }, |
| { |
| "batch_accuracy": 0.8392857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 248.25447845458984, |
| "epoch": 69.13793103448276, |
| "grad_norm": 0.41907478180367097, |
| "kl": 0.1656494140625, |
| "learning_rate": 1e-06, |
| "loss": 0.005, |
| "reward": 0.839285746216774, |
| "reward_std": 0.05050762742757797, |
| "rewards/unified_reward_func": 0.839285746216774, |
| "step": 415 |
| }, |
| { |
| "clip_ratio": 0.0014105399022810161, |
| "epoch": 69.27586206896552, |
| "grad_norm": 0.25769744775290754, |
| "kl": 0.183837890625, |
| "learning_rate": 1e-06, |
| "loss": 0.0043, |
| "step": 416 |
| }, |
| { |
| "batch_accuracy": 0.8125, |
| "clip_ratio": 0.0, |
| "completion_length": 237.3348388671875, |
| "epoch": 69.41379310344827, |
| "grad_norm": 0.189936523053462, |
| "kl": 0.169921875, |
| "learning_rate": 1e-06, |
| "loss": -0.0057, |
| "reward": 0.8125000298023224, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8125000298023224, |
| "step": 417 |
| }, |
| { |
| "clip_ratio": 0.00024228040274465457, |
| "epoch": 69.55172413793103, |
| "grad_norm": 0.14852661920422203, |
| "kl": 0.16748046875, |
| "learning_rate": 1e-06, |
| "loss": -0.0061, |
| "step": 418 |
| }, |
| { |
| "batch_accuracy": 0.8794642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 240.67412185668945, |
| "epoch": 69.6896551724138, |
| "grad_norm": 0.30067919176795704, |
| "kl": 0.1591796875, |
| "learning_rate": 1e-06, |
| "loss": 0.0019, |
| "reward": 0.8794643133878708, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.8794643133878708, |
| "step": 419 |
| }, |
| { |
| "clip_ratio": 0.0004018953113700263, |
| "epoch": 69.82758620689656, |
| "grad_norm": 0.19383722712493728, |
| "kl": 0.16796875, |
| "learning_rate": 1e-06, |
| "loss": 0.0015, |
| "step": 420 |
| }, |
| { |
| "batch_accuracy": 0.7678571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 224.6696548461914, |
| "epoch": 70.13793103448276, |
| "grad_norm": 0.6500836334692476, |
| "kl": 0.2109375, |
| "learning_rate": 1e-06, |
| "loss": -0.004, |
| "reward": 0.7678571790456772, |
| "reward_std": 0.05050762742757797, |
| "rewards/unified_reward_func": 0.7678571790456772, |
| "step": 421 |
| }, |
| { |
| "clip_ratio": 0.00046241712698247284, |
| "epoch": 70.27586206896552, |
| "grad_norm": 0.38904128273042526, |
| "kl": 0.204345703125, |
| "learning_rate": 1e-06, |
| "loss": -0.005, |
| "step": 422 |
| }, |
| { |
| "batch_accuracy": 0.9553571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 212.03125762939453, |
| "epoch": 70.41379310344827, |
| "grad_norm": 0.5739109252892366, |
| "kl": 0.4423828125, |
| "learning_rate": 1e-06, |
| "loss": -0.0116, |
| "reward": 0.9553571492433548, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9553571492433548, |
| "step": 423 |
| }, |
| { |
| "clip_ratio": 8.943354623625055e-05, |
| "epoch": 70.55172413793103, |
| "grad_norm": 0.16299775366551209, |
| "kl": 0.15478515625, |
| "learning_rate": 1e-06, |
| "loss": -0.012, |
| "step": 424 |
| }, |
| { |
| "batch_accuracy": 0.8794642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 228.59375762939453, |
| "epoch": 70.6896551724138, |
| "grad_norm": 0.38581004341045927, |
| "kl": 0.68408203125, |
| "learning_rate": 1e-06, |
| "loss": 0.0007, |
| "reward": 0.8794642984867096, |
| "reward_std": 0.018483899533748627, |
| "rewards/unified_reward_func": 0.8794642984867096, |
| "step": 425 |
| }, |
| { |
| "clip_ratio": 0.0001823043276090175, |
| "epoch": 70.82758620689656, |
| "grad_norm": 2.0528087697459636, |
| "kl": 0.314453125, |
| "learning_rate": 1e-06, |
| "loss": 0.0015, |
| "step": 426 |
| }, |
| { |
| "batch_accuracy": 0.8214285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 242.22768783569336, |
| "epoch": 71.13793103448276, |
| "grad_norm": 0.04625840730424008, |
| "kl": 0.23095703125, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "reward": 0.8214286267757416, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.8214286267757416, |
| "step": 427 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 71.27586206896552, |
| "grad_norm": 0.03143840003558422, |
| "kl": 0.2021484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 428 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 233.1562614440918, |
| "epoch": 71.41379310344827, |
| "grad_norm": 0.5009023723263193, |
| "kl": 0.207763671875, |
| "learning_rate": 1e-06, |
| "loss": 0.003, |
| "reward": 0.8482143431901932, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8482143431901932, |
| "step": 429 |
| }, |
| { |
| "clip_ratio": 0.0003354004038556013, |
| "epoch": 71.55172413793103, |
| "grad_norm": 0.2538041516707458, |
| "kl": 0.1826171875, |
| "learning_rate": 1e-06, |
| "loss": 0.0022, |
| "step": 430 |
| }, |
| { |
| "batch_accuracy": 0.9151785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 222.32143783569336, |
| "epoch": 71.6896551724138, |
| "grad_norm": 0.6051502407175923, |
| "kl": 0.1630859375, |
| "learning_rate": 1e-06, |
| "loss": 0.0089, |
| "reward": 0.9151785969734192, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.9151785969734192, |
| "step": 431 |
| }, |
| { |
| "clip_ratio": 0.0003940899041481316, |
| "epoch": 71.82758620689656, |
| "grad_norm": 0.2518463742920773, |
| "kl": 0.155517578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0084, |
| "step": 432 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 225.35268783569336, |
| "epoch": 72.13793103448276, |
| "grad_norm": 2.3905013205508494, |
| "kl": 0.99169921875, |
| "learning_rate": 1e-06, |
| "loss": 0.001, |
| "reward": 0.8482143431901932, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8482143431901932, |
| "step": 433 |
| }, |
| { |
| "clip_ratio": 0.0006973157869651914, |
| "epoch": 72.27586206896552, |
| "grad_norm": 0.23797546293977923, |
| "kl": 0.26318359375, |
| "learning_rate": 1e-06, |
| "loss": 0.0003, |
| "step": 434 |
| }, |
| { |
| "batch_accuracy": 0.84375, |
| "clip_ratio": 0.0, |
| "completion_length": 217.71875762939453, |
| "epoch": 72.41379310344827, |
| "grad_norm": 0.7831567380243376, |
| "kl": 0.1953125, |
| "learning_rate": 1e-06, |
| "loss": 0.002, |
| "reward": 0.8437500298023224, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.8437500298023224, |
| "step": 435 |
| }, |
| { |
| "clip_ratio": 0.0015309452719520777, |
| "epoch": 72.55172413793103, |
| "grad_norm": 0.34815002643262916, |
| "kl": 0.27490234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0013, |
| "step": 436 |
| }, |
| { |
| "batch_accuracy": 0.9196428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 218.4241180419922, |
| "epoch": 72.6896551724138, |
| "grad_norm": 0.398082329113118, |
| "kl": 0.168701171875, |
| "learning_rate": 1e-06, |
| "loss": -0.0058, |
| "reward": 0.9196428656578064, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9196428656578064, |
| "step": 437 |
| }, |
| { |
| "clip_ratio": 0.00032450424623675644, |
| "epoch": 72.82758620689656, |
| "grad_norm": 0.20485368896653353, |
| "kl": 0.172119140625, |
| "learning_rate": 1e-06, |
| "loss": -0.0062, |
| "step": 438 |
| }, |
| { |
| "batch_accuracy": 0.8169642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 211.16965103149414, |
| "epoch": 73.13793103448276, |
| "grad_norm": 0.33955964196499566, |
| "kl": 0.20263671875, |
| "learning_rate": 1e-06, |
| "loss": 0.0007, |
| "reward": 0.8169643133878708, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.8169643133878708, |
| "step": 439 |
| }, |
| { |
| "clip_ratio": 2.829975164786447e-05, |
| "epoch": 73.27586206896552, |
| "grad_norm": 0.26818712559841884, |
| "kl": 0.270751953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0003, |
| "step": 440 |
| }, |
| { |
| "batch_accuracy": 0.8794642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 245.52680206298828, |
| "epoch": 73.41379310344827, |
| "grad_norm": 0.6770300911622293, |
| "kl": 0.58935546875, |
| "learning_rate": 1e-06, |
| "loss": 0.005, |
| "reward": 0.8794643133878708, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.8794643133878708, |
| "step": 441 |
| }, |
| { |
| "clip_ratio": 0.00035833044967148453, |
| "epoch": 73.55172413793103, |
| "grad_norm": 79.74265910678565, |
| "kl": 35.379638671875, |
| "learning_rate": 1e-06, |
| "loss": 0.0388, |
| "step": 442 |
| }, |
| { |
| "batch_accuracy": 0.9241071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 220.05358123779297, |
| "epoch": 73.6896551724138, |
| "grad_norm": 1.2885073773349964, |
| "kl": 1.6875, |
| "learning_rate": 1e-06, |
| "loss": 0.0027, |
| "reward": 0.924107164144516, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.924107164144516, |
| "step": 443 |
| }, |
| { |
| "clip_ratio": 5.87406029808335e-05, |
| "epoch": 73.82758620689656, |
| "grad_norm": 0.23799297864282562, |
| "kl": 0.529541015625, |
| "learning_rate": 1e-06, |
| "loss": 0.0014, |
| "step": 444 |
| }, |
| { |
| "batch_accuracy": 0.9151785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 244.63394165039062, |
| "epoch": 74.13793103448276, |
| "grad_norm": 0.40842932428295264, |
| "kl": 0.1495361328125, |
| "learning_rate": 1e-06, |
| "loss": -0.0062, |
| "reward": 0.915178582072258, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.915178582072258, |
| "step": 445 |
| }, |
| { |
| "clip_ratio": 0.0008320850902236998, |
| "epoch": 74.27586206896552, |
| "grad_norm": 0.2382784575121656, |
| "kl": 0.1640625, |
| "learning_rate": 1e-06, |
| "loss": -0.0067, |
| "step": 446 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 230.9866180419922, |
| "epoch": 74.41379310344827, |
| "grad_norm": 0.1997408281341476, |
| "kl": 0.208740234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0018, |
| "reward": 0.8482143133878708, |
| "reward_std": 0.016532503068447113, |
| "rewards/unified_reward_func": 0.8482143133878708, |
| "step": 447 |
| }, |
| { |
| "clip_ratio": 0.000311763898935169, |
| "epoch": 74.55172413793103, |
| "grad_norm": 0.1335107548690185, |
| "kl": 0.16796875, |
| "learning_rate": 1e-06, |
| "loss": 0.0015, |
| "step": 448 |
| }, |
| { |
| "batch_accuracy": 0.8928571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 222.02679824829102, |
| "epoch": 74.6896551724138, |
| "grad_norm": 0.10165897354603974, |
| "kl": 0.28369140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0003, |
| "reward": 0.892857164144516, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.892857164144516, |
| "step": 449 |
| }, |
| { |
| "epoch": 74.82758620689656, |
| "grad_norm": 0.02722702905999937, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 450 |
| }, |
| { |
| "epoch": 74.82758620689656, |
| "eval_batch_accuracy": 0.7607142857142858, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 239.97046305338543, |
| "eval_kl": 0.4384765625, |
| "eval_loss": 0.010263421572744846, |
| "eval_reward": 0.760714324315389, |
| "eval_reward_std": 0.10253030310074489, |
| "eval_rewards/unified_reward_func": 0.760714324315389, |
| "eval_runtime": 292.1226, |
| "eval_samples_per_second": 0.342, |
| "eval_steps_per_second": 0.007, |
| "step": 450 |
| }, |
| { |
| "batch_accuracy": 0.7767857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 292.33483123779297, |
| "epoch": 75.13793103448276, |
| "grad_norm": 1.2280644592028591, |
| "kl": 0.53375244140625, |
| "learning_rate": 1e-06, |
| "loss": -0.0072, |
| "reward": 0.776785746216774, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.776785746216774, |
| "step": 451 |
| }, |
| { |
| "clip_ratio": 0.00022370762599166483, |
| "epoch": 75.27586206896552, |
| "grad_norm": 3294.426305780333, |
| "kl": 0.3575439453125, |
| "learning_rate": 1e-06, |
| "loss": 1.0297, |
| "step": 452 |
| }, |
| { |
| "batch_accuracy": 0.9642857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 210.72322463989258, |
| "epoch": 75.41379310344827, |
| "grad_norm": 0.012716440951391888, |
| "kl": 0.1534423828125, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "reward": 0.9642857313156128, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.9642857313156128, |
| "step": 453 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 75.55172413793103, |
| "grad_norm": 0.012240949557013305, |
| "kl": 0.1512451171875, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 454 |
| }, |
| { |
| "batch_accuracy": 0.9196428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 209.94643783569336, |
| "epoch": 75.6896551724138, |
| "grad_norm": 0.5247671039188904, |
| "kl": 0.160400390625, |
| "learning_rate": 1e-06, |
| "loss": 0.0076, |
| "reward": 0.9196429252624512, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9196429252624512, |
| "step": 455 |
| }, |
| { |
| "clip_ratio": 0.00023868290008977056, |
| "epoch": 75.82758620689656, |
| "grad_norm": 0.2792572669606453, |
| "kl": 0.161865234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0069, |
| "step": 456 |
| }, |
| { |
| "batch_accuracy": 0.8571428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 188.30804443359375, |
| "epoch": 76.13793103448276, |
| "grad_norm": 0.04946528678616401, |
| "kl": 0.218017578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "reward": 0.8571428656578064, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.8571428656578064, |
| "step": 457 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 76.27586206896552, |
| "grad_norm": 0.039462119096367015, |
| "kl": 0.21142578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 458 |
| }, |
| { |
| "batch_accuracy": 0.8928571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 229.00893783569336, |
| "epoch": 76.41379310344827, |
| "grad_norm": 0.5075704410401828, |
| "kl": 0.501220703125, |
| "learning_rate": 1e-06, |
| "loss": 0.0005, |
| "reward": 0.8928571939468384, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.8928571939468384, |
| "step": 459 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 76.55172413793103, |
| "grad_norm": 0.03775215604367821, |
| "kl": 0.20849609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 460 |
| }, |
| { |
| "batch_accuracy": 0.8616071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 254.00000762939453, |
| "epoch": 76.6896551724138, |
| "grad_norm": 0.7803983058510199, |
| "kl": 0.3798828125, |
| "learning_rate": 1e-06, |
| "loss": 0.0166, |
| "reward": 0.861607164144516, |
| "reward_std": 0.07966703735291958, |
| "rewards/unified_reward_func": 0.861607164144516, |
| "step": 461 |
| }, |
| { |
| "clip_ratio": 0.0014210399895091541, |
| "epoch": 76.82758620689656, |
| "grad_norm": 0.49834308317287584, |
| "kl": 0.298583984375, |
| "learning_rate": 1e-06, |
| "loss": 0.0151, |
| "step": 462 |
| }, |
| { |
| "batch_accuracy": 0.8392857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 242.5759048461914, |
| "epoch": 77.13793103448276, |
| "grad_norm": 0.5803647637709338, |
| "kl": 0.32666015625, |
| "learning_rate": 1e-06, |
| "loss": -0.0033, |
| "reward": 0.8392857313156128, |
| "reward_std": 0.05050762742757797, |
| "rewards/unified_reward_func": 0.8392857313156128, |
| "step": 463 |
| }, |
| { |
| "clip_ratio": 0.0003732922159542795, |
| "epoch": 77.27586206896552, |
| "grad_norm": 0.37156821061329787, |
| "kl": 0.266357421875, |
| "learning_rate": 1e-06, |
| "loss": -0.0041, |
| "step": 464 |
| }, |
| { |
| "batch_accuracy": 0.8392857142857142, |
| "clip_ratio": 0.0, |
| "completion_length": 227.93304443359375, |
| "epoch": 77.41379310344827, |
| "grad_norm": 0.5552624609943606, |
| "kl": 0.172119140625, |
| "learning_rate": 1e-06, |
| "loss": -0.0093, |
| "reward": 0.8392857611179352, |
| "reward_std": 0.05050762742757797, |
| "rewards/unified_reward_func": 0.8392857611179352, |
| "step": 465 |
| }, |
| { |
| "clip_ratio": 0.0005572987865889445, |
| "epoch": 77.55172413793103, |
| "grad_norm": 0.29331584969110686, |
| "kl": 0.181640625, |
| "learning_rate": 1e-06, |
| "loss": -0.0101, |
| "step": 466 |
| }, |
| { |
| "batch_accuracy": 0.9508928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 206.97768783569336, |
| "epoch": 77.6896551724138, |
| "grad_norm": 0.4342585602914251, |
| "kl": 0.2197265625, |
| "learning_rate": 1e-06, |
| "loss": 0.0008, |
| "reward": 0.9508928954601288, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.9508928954601288, |
| "step": 467 |
| }, |
| { |
| "clip_ratio": 0.0007359530427493155, |
| "epoch": 77.82758620689656, |
| "grad_norm": 0.9475669204813898, |
| "kl": 0.556640625, |
| "learning_rate": 1e-06, |
| "loss": 0.0006, |
| "step": 468 |
| }, |
| { |
| "batch_accuracy": 0.7723214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 252.02233123779297, |
| "epoch": 78.13793103448276, |
| "grad_norm": 0.6308618189250425, |
| "kl": 0.570556640625, |
| "learning_rate": 1e-06, |
| "loss": 0.0042, |
| "reward": 0.7723214626312256, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.7723214626312256, |
| "step": 469 |
| }, |
| { |
| "clip_ratio": 0.0002528606928535737, |
| "epoch": 78.27586206896552, |
| "grad_norm": 0.3335563972152877, |
| "kl": 0.587158203125, |
| "learning_rate": 1e-06, |
| "loss": 0.0032, |
| "step": 470 |
| }, |
| { |
| "batch_accuracy": 0.9196428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 234.14286041259766, |
| "epoch": 78.41379310344827, |
| "grad_norm": 0.4791093710644942, |
| "kl": 0.271728515625, |
| "learning_rate": 1e-06, |
| "loss": 0.0074, |
| "reward": 0.9196428656578064, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9196428656578064, |
| "step": 471 |
| }, |
| { |
| "clip_ratio": 0.00046849474165355787, |
| "epoch": 78.55172413793103, |
| "grad_norm": 0.28861887009948634, |
| "kl": 0.249267578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0068, |
| "step": 472 |
| }, |
| { |
| "batch_accuracy": 0.9285714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 206.95983123779297, |
| "epoch": 78.6896551724138, |
| "grad_norm": 0.13724663375532223, |
| "kl": 0.2587890625, |
| "learning_rate": 1e-06, |
| "loss": 0.0003, |
| "reward": 0.9285714328289032, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.9285714328289032, |
| "step": 473 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 78.82758620689656, |
| "grad_norm": 0.037068066277568514, |
| "kl": 0.216064453125, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 474 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 199.35715103149414, |
| "epoch": 79.13793103448276, |
| "grad_norm": 0.3264825065714323, |
| "kl": 0.2763671875, |
| "learning_rate": 1e-06, |
| "loss": -0.009, |
| "reward": 0.848214328289032, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.848214328289032, |
| "step": 475 |
| }, |
| { |
| "clip_ratio": 0.0003418240521568805, |
| "epoch": 79.27586206896552, |
| "grad_norm": 0.18676471981083279, |
| "kl": 0.2568359375, |
| "learning_rate": 1e-06, |
| "loss": -0.0093, |
| "step": 476 |
| }, |
| { |
| "batch_accuracy": 0.8883928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 233.7946548461914, |
| "epoch": 79.41379310344827, |
| "grad_norm": 1.101668515839696, |
| "kl": 0.56103515625, |
| "learning_rate": 1e-06, |
| "loss": -0.0019, |
| "reward": 0.8883928954601288, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.8883928954601288, |
| "step": 477 |
| }, |
| { |
| "clip_ratio": 0.00017312313138972968, |
| "epoch": 79.55172413793103, |
| "grad_norm": 0.1583937351723072, |
| "kl": 0.229248046875, |
| "learning_rate": 1e-06, |
| "loss": -0.0022, |
| "step": 478 |
| }, |
| { |
| "batch_accuracy": 0.8035714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 217.97322463989258, |
| "epoch": 79.6896551724138, |
| "grad_norm": 0.9780994951787898, |
| "kl": 0.823974609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0107, |
| "reward": 0.8035714626312256, |
| "reward_std": 0.05050762742757797, |
| "rewards/unified_reward_func": 0.8035714626312256, |
| "step": 479 |
| }, |
| { |
| "clip_ratio": 0.0002679546087165363, |
| "epoch": 79.82758620689656, |
| "grad_norm": 0.4882068912890497, |
| "kl": 0.30224609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0094, |
| "step": 480 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 234.30805206298828, |
| "epoch": 80.13793103448276, |
| "grad_norm": 0.25335686041912453, |
| "kl": 0.414306640625, |
| "learning_rate": 1e-06, |
| "loss": -0.0248, |
| "reward": 0.8839285969734192, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839285969734192, |
| "step": 481 |
| }, |
| { |
| "clip_ratio": 8.276415883301524e-05, |
| "epoch": 80.27586206896552, |
| "grad_norm": 0.11536539576915139, |
| "kl": 0.22021484375, |
| "learning_rate": 1e-06, |
| "loss": -0.0252, |
| "step": 482 |
| }, |
| { |
| "batch_accuracy": 0.7857142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 209.30804061889648, |
| "epoch": 80.41379310344827, |
| "grad_norm": 0.11998413938965553, |
| "kl": 0.2314453125, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "reward": 0.785714328289032, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.785714328289032, |
| "step": 483 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 80.55172413793103, |
| "grad_norm": 0.07272808970933248, |
| "kl": 0.202392578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 484 |
| }, |
| { |
| "batch_accuracy": 0.9151785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 220.6696548461914, |
| "epoch": 80.6896551724138, |
| "grad_norm": 0.20330006705789813, |
| "kl": 0.232421875, |
| "learning_rate": 1e-06, |
| "loss": -0.0026, |
| "reward": 0.9151786118745804, |
| "reward_std": 0.018483899533748627, |
| "rewards/unified_reward_func": 0.9151786118745804, |
| "step": 485 |
| }, |
| { |
| "clip_ratio": 8.486562728649005e-05, |
| "epoch": 80.82758620689656, |
| "grad_norm": 0.13781009873072228, |
| "kl": 0.181640625, |
| "learning_rate": 1e-06, |
| "loss": -0.0029, |
| "step": 486 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 207.99108123779297, |
| "epoch": 81.13793103448276, |
| "grad_norm": 0.4285326309410844, |
| "kl": 0.1611328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0008, |
| "reward": 0.8839286267757416, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839286267757416, |
| "step": 487 |
| }, |
| { |
| "clip_ratio": 0.0002343350297451252, |
| "epoch": 81.27586206896552, |
| "grad_norm": 0.20850823605286758, |
| "kl": 0.1688232421875, |
| "learning_rate": 1e-06, |
| "loss": 0.0004, |
| "step": 488 |
| }, |
| { |
| "batch_accuracy": 0.9241071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 216.19644165039062, |
| "epoch": 81.41379310344827, |
| "grad_norm": 0.30122462050916987, |
| "kl": 0.1904296875, |
| "learning_rate": 1e-06, |
| "loss": -0.0023, |
| "reward": 0.924107164144516, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.924107164144516, |
| "step": 489 |
| }, |
| { |
| "clip_ratio": 8.710801193956286e-05, |
| "epoch": 81.55172413793103, |
| "grad_norm": 1.4111707805241815, |
| "kl": 0.67236328125, |
| "learning_rate": 1e-06, |
| "loss": -0.0021, |
| "step": 490 |
| }, |
| { |
| "batch_accuracy": 0.7678571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 244.08037185668945, |
| "epoch": 81.6896551724138, |
| "grad_norm": 0.4743000711442227, |
| "kl": 0.30810546875, |
| "learning_rate": 1e-06, |
| "loss": -0.0068, |
| "reward": 0.7678571790456772, |
| "reward_std": 0.033065006136894226, |
| "rewards/unified_reward_func": 0.7678571790456772, |
| "step": 491 |
| }, |
| { |
| "clip_ratio": 0.00022754022211302072, |
| "epoch": 81.82758620689656, |
| "grad_norm": 0.26704613852928916, |
| "kl": 0.24658203125, |
| "learning_rate": 1e-06, |
| "loss": -0.0074, |
| "step": 492 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 246.3928680419922, |
| "epoch": 82.13793103448276, |
| "grad_norm": 0.5254256464673122, |
| "kl": 0.374267578125, |
| "learning_rate": 1e-06, |
| "loss": -0.0061, |
| "reward": 0.8839285969734192, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839285969734192, |
| "step": 493 |
| }, |
| { |
| "clip_ratio": 0.0004009578133263858, |
| "epoch": 82.27586206896552, |
| "grad_norm": 0.2680176309600896, |
| "kl": 0.301513671875, |
| "learning_rate": 1e-06, |
| "loss": -0.0067, |
| "step": 494 |
| }, |
| { |
| "batch_accuracy": 0.8928571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 214.97768783569336, |
| "epoch": 82.41379310344827, |
| "grad_norm": 0.7978444005747602, |
| "kl": 0.912353515625, |
| "learning_rate": 1e-06, |
| "loss": 0.0009, |
| "reward": 0.8928571939468384, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.8928571939468384, |
| "step": 495 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 82.55172413793103, |
| "grad_norm": 0.025686755334383627, |
| "kl": 0.19580078125, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 496 |
| }, |
| { |
| "batch_accuracy": 0.8169642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 236.85269165039062, |
| "epoch": 82.6896551724138, |
| "grad_norm": 0.11684909474736647, |
| "kl": 0.25927734375, |
| "learning_rate": 1e-06, |
| "loss": 0.0007, |
| "reward": 0.816964328289032, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.816964328289032, |
| "step": 497 |
| }, |
| { |
| "clip_ratio": 3.523856503306888e-05, |
| "epoch": 82.82758620689656, |
| "grad_norm": 0.06796555451043418, |
| "kl": 0.19921875, |
| "learning_rate": 1e-06, |
| "loss": 0.0005, |
| "step": 498 |
| }, |
| { |
| "batch_accuracy": 0.7276785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 245.62054443359375, |
| "epoch": 83.13793103448276, |
| "grad_norm": 0.8222045042128722, |
| "kl": 0.2041015625, |
| "learning_rate": 1e-06, |
| "loss": 0.0066, |
| "reward": 0.7276785969734192, |
| "reward_std": 0.06313453428447247, |
| "rewards/unified_reward_func": 0.7276785969734192, |
| "step": 499 |
| }, |
| { |
| "epoch": 83.27586206896552, |
| "grad_norm": 0.5300778222275132, |
| "learning_rate": 1e-06, |
| "loss": 0.0054, |
| "step": 500 |
| }, |
| { |
| "epoch": 83.27586206896552, |
| "eval_batch_accuracy": 0.7404761904761905, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 242.2316131591797, |
| "eval_kl": 0.205615234375, |
| "eval_loss": -0.00046114774886518717, |
| "eval_reward": 0.7404762228329976, |
| "eval_reward_std": 0.11178621302048365, |
| "eval_rewards/unified_reward_func": 0.7404762228329976, |
| "eval_runtime": 284.5712, |
| "eval_samples_per_second": 0.351, |
| "eval_steps_per_second": 0.007, |
| "step": 500 |
| }, |
| { |
| "batch_accuracy": 0.9553571428571428, |
| "clip_ratio": 0.0003930836610379629, |
| "completion_length": 223.55358123779297, |
| "epoch": 83.41379310344827, |
| "grad_norm": 0.3338264355322204, |
| "kl": 0.15478515625, |
| "learning_rate": 1e-06, |
| "loss": -0.0017, |
| "reward": 0.9553571939468384, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9553571939468384, |
| "step": 501 |
| }, |
| { |
| "clip_ratio": 0.0004324326873756945, |
| "epoch": 83.55172413793103, |
| "grad_norm": 0.18037033885748482, |
| "kl": 0.1173095703125, |
| "learning_rate": 1e-06, |
| "loss": -0.0021, |
| "step": 502 |
| }, |
| { |
| "batch_accuracy": 0.8883928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 233.3437614440918, |
| "epoch": 83.6896551724138, |
| "grad_norm": 0.24428580583668974, |
| "kl": 0.1396484375, |
| "learning_rate": 1e-06, |
| "loss": -0.0004, |
| "reward": 0.8883928805589676, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.8883928805589676, |
| "step": 503 |
| }, |
| { |
| "clip_ratio": 0.00011091393389506266, |
| "epoch": 83.82758620689656, |
| "grad_norm": 0.12753916151633782, |
| "kl": 0.1373291015625, |
| "learning_rate": 1e-06, |
| "loss": -0.0006, |
| "step": 504 |
| }, |
| { |
| "batch_accuracy": 0.8928571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 217.25447463989258, |
| "epoch": 84.13793103448276, |
| "grad_norm": 0.03625836028576517, |
| "kl": 0.1346435546875, |
| "learning_rate": 1e-06, |
| "loss": 0.0001, |
| "reward": 0.892857164144516, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.892857164144516, |
| "step": 505 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 84.27586206896552, |
| "grad_norm": 0.022544787709227847, |
| "kl": 0.109130859375, |
| "learning_rate": 1e-06, |
| "loss": 0.0001, |
| "step": 506 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 238.36161422729492, |
| "epoch": 84.41379310344827, |
| "grad_norm": 0.7982539820094933, |
| "kl": 0.2203369140625, |
| "learning_rate": 1e-06, |
| "loss": -0.0081, |
| "reward": 0.8482143431901932, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8482143431901932, |
| "step": 507 |
| }, |
| { |
| "clip_ratio": 0.00042628360097296536, |
| "epoch": 84.55172413793103, |
| "grad_norm": 0.3511699215249297, |
| "kl": 0.2071533203125, |
| "learning_rate": 1e-06, |
| "loss": -0.0089, |
| "step": 508 |
| }, |
| { |
| "batch_accuracy": 0.8928571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 260.40625381469727, |
| "epoch": 84.6896551724138, |
| "grad_norm": 0.011861750126758629, |
| "kl": 0.08392333984375, |
| "learning_rate": 1e-06, |
| "loss": 0.0001, |
| "reward": 0.8928571939468384, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.8928571939468384, |
| "step": 509 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 84.82758620689656, |
| "grad_norm": 0.0115087349123355, |
| "kl": 0.08526611328125, |
| "learning_rate": 1e-06, |
| "loss": 0.0001, |
| "step": 510 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 214.8973274230957, |
| "epoch": 85.13793103448276, |
| "grad_norm": 0.36929124899796995, |
| "kl": 0.1583251953125, |
| "learning_rate": 1e-06, |
| "loss": -0.0044, |
| "reward": 0.8839286118745804, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839286118745804, |
| "step": 511 |
| }, |
| { |
| "clip_ratio": 0.00018426612950861454, |
| "epoch": 85.27586206896552, |
| "grad_norm": 0.15182488091269972, |
| "kl": 0.1295166015625, |
| "learning_rate": 1e-06, |
| "loss": -0.0049, |
| "step": 512 |
| }, |
| { |
| "batch_accuracy": 0.9508928571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 234.2321548461914, |
| "epoch": 85.41379310344827, |
| "grad_norm": 0.3553720690267824, |
| "kl": 0.091796875, |
| "learning_rate": 1e-06, |
| "loss": -0.0045, |
| "reward": 0.9508928656578064, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.9508928656578064, |
| "step": 513 |
| }, |
| { |
| "clip_ratio": 0.00010698674213927006, |
| "epoch": 85.55172413793103, |
| "grad_norm": 0.21153475186501686, |
| "kl": 0.114990234375, |
| "learning_rate": 1e-06, |
| "loss": -0.005, |
| "step": 514 |
| }, |
| { |
| "batch_accuracy": 0.8794642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 261.16965103149414, |
| "epoch": 85.6896551724138, |
| "grad_norm": 0.4148941424286881, |
| "kl": 0.1607666015625, |
| "learning_rate": 1e-06, |
| "loss": -0.0028, |
| "reward": 0.879464328289032, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.879464328289032, |
| "step": 515 |
| }, |
| { |
| "clip_ratio": 0.0002979067139676772, |
| "epoch": 85.82758620689656, |
| "grad_norm": 0.2795928071224825, |
| "kl": 0.1551513671875, |
| "learning_rate": 1e-06, |
| "loss": -0.0035, |
| "step": 516 |
| }, |
| { |
| "batch_accuracy": 0.84375, |
| "clip_ratio": 0.0, |
| "completion_length": 253.6428680419922, |
| "epoch": 86.13793103448276, |
| "grad_norm": 0.27804716865753637, |
| "kl": 0.17291259765625, |
| "learning_rate": 1e-06, |
| "loss": -0.0075, |
| "reward": 0.8437500447034836, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.8437500447034836, |
| "step": 517 |
| }, |
| { |
| "clip_ratio": 0.0001618318710825406, |
| "epoch": 86.27586206896552, |
| "grad_norm": 0.20800322960868664, |
| "kl": 0.203125, |
| "learning_rate": 1e-06, |
| "loss": -0.0078, |
| "step": 518 |
| }, |
| { |
| "batch_accuracy": 0.9285714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 245.8303680419922, |
| "epoch": 86.41379310344827, |
| "grad_norm": 0.07559120994905119, |
| "kl": 0.1693115234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "reward": 0.9285714328289032, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.9285714328289032, |
| "step": 519 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 86.55172413793103, |
| "grad_norm": 0.027510075620572266, |
| "kl": 0.1263427734375, |
| "learning_rate": 1e-06, |
| "loss": 0.0001, |
| "step": 520 |
| }, |
| { |
| "batch_accuracy": 0.8080357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 257.6651954650879, |
| "epoch": 86.6896551724138, |
| "grad_norm": 0.317410842856532, |
| "kl": 0.18505859375, |
| "learning_rate": 1e-06, |
| "loss": -0.0014, |
| "reward": 0.808035746216774, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.808035746216774, |
| "step": 521 |
| }, |
| { |
| "clip_ratio": 0.00045395906636258587, |
| "epoch": 86.82758620689656, |
| "grad_norm": 0.1853671737450051, |
| "kl": 0.19140625, |
| "learning_rate": 1e-06, |
| "loss": -0.0018, |
| "step": 522 |
| }, |
| { |
| "batch_accuracy": 0.8214285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 237.3259048461914, |
| "epoch": 87.13793103448276, |
| "grad_norm": 0.0430564348245684, |
| "kl": 0.2119140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "reward": 0.821428582072258, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.821428582072258, |
| "step": 523 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 87.27586206896552, |
| "grad_norm": 0.034899739644111305, |
| "kl": 0.1956787109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 524 |
| }, |
| { |
| "batch_accuracy": 0.8035714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 268.3705520629883, |
| "epoch": 87.41379310344827, |
| "grad_norm": 0.4141970723107917, |
| "kl": 0.1942138671875, |
| "learning_rate": 1e-06, |
| "loss": -0.0078, |
| "reward": 0.8035714626312256, |
| "reward_std": 0.05050762742757797, |
| "rewards/unified_reward_func": 0.8035714626312256, |
| "step": 525 |
| }, |
| { |
| "clip_ratio": 0.0006828630139352754, |
| "epoch": 87.55172413793103, |
| "grad_norm": 0.32788510076339705, |
| "kl": 0.2764892578125, |
| "learning_rate": 1e-06, |
| "loss": -0.0084, |
| "step": 526 |
| }, |
| { |
| "batch_accuracy": 0.9553571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 250.99108505249023, |
| "epoch": 87.6896551724138, |
| "grad_norm": 0.22739404521765016, |
| "kl": 0.126953125, |
| "learning_rate": 1e-06, |
| "loss": -0.004, |
| "reward": 0.9553571939468384, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9553571939468384, |
| "step": 527 |
| }, |
| { |
| "clip_ratio": 0.00020626326295314357, |
| "epoch": 87.82758620689656, |
| "grad_norm": 0.14809417054594567, |
| "kl": 0.1229248046875, |
| "learning_rate": 1e-06, |
| "loss": -0.0043, |
| "step": 528 |
| }, |
| { |
| "batch_accuracy": 0.8482142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 286.08484268188477, |
| "epoch": 88.13793103448276, |
| "grad_norm": 0.36924668148503553, |
| "kl": 0.1690673828125, |
| "learning_rate": 1e-06, |
| "loss": -0.0093, |
| "reward": 0.8482142984867096, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8482142984867096, |
| "step": 529 |
| }, |
| { |
| "clip_ratio": 0.0002911818137363298, |
| "epoch": 88.27586206896552, |
| "grad_norm": 0.24608301664562246, |
| "kl": 0.1900634765625, |
| "learning_rate": 1e-06, |
| "loss": -0.0098, |
| "step": 530 |
| }, |
| { |
| "batch_accuracy": 0.8571428571428572, |
| "clip_ratio": 0.0, |
| "completion_length": 227.38393783569336, |
| "epoch": 88.41379310344827, |
| "grad_norm": 0.3658054197303534, |
| "kl": 0.669677734375, |
| "learning_rate": 1e-06, |
| "loss": 0.0007, |
| "reward": 0.8571428656578064, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.8571428656578064, |
| "step": 531 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 88.55172413793103, |
| "grad_norm": 0.07118533950169958, |
| "kl": 0.24267578125, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 532 |
| }, |
| { |
| "batch_accuracy": 0.84375, |
| "clip_ratio": 0.0, |
| "completion_length": 245.21430206298828, |
| "epoch": 88.6896551724138, |
| "grad_norm": 0.4255380426635142, |
| "kl": 0.1558837890625, |
| "learning_rate": 1e-06, |
| "loss": -0.0099, |
| "reward": 0.8437500298023224, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.8437500298023224, |
| "step": 533 |
| }, |
| { |
| "clip_ratio": 0.0009154944564215839, |
| "epoch": 88.82758620689656, |
| "grad_norm": 0.25713411225878763, |
| "kl": 0.162841796875, |
| "learning_rate": 1e-06, |
| "loss": -0.0106, |
| "step": 534 |
| }, |
| { |
| "batch_accuracy": 0.8883928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 273.3437614440918, |
| "epoch": 89.13793103448276, |
| "grad_norm": 0.24874726271726094, |
| "kl": 0.154541015625, |
| "learning_rate": 1e-06, |
| "loss": 0.0008, |
| "reward": 0.8883928954601288, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.8883928954601288, |
| "step": 535 |
| }, |
| { |
| "clip_ratio": 0.000315327662974596, |
| "epoch": 89.27586206896552, |
| "grad_norm": 0.15339425804522938, |
| "kl": 0.2099609375, |
| "learning_rate": 1e-06, |
| "loss": 0.0007, |
| "step": 536 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 223.38394165039062, |
| "epoch": 89.41379310344827, |
| "grad_norm": 0.43792988145907863, |
| "kl": 0.3629150390625, |
| "learning_rate": 1e-06, |
| "loss": -0.0038, |
| "reward": 0.8839286267757416, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839286267757416, |
| "step": 537 |
| }, |
| { |
| "clip_ratio": 0.0004368049849290401, |
| "epoch": 89.55172413793103, |
| "grad_norm": 0.22254866311488028, |
| "kl": 0.355224609375, |
| "learning_rate": 1e-06, |
| "loss": -0.0042, |
| "step": 538 |
| }, |
| { |
| "batch_accuracy": 0.7723214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 285.77233123779297, |
| "epoch": 89.6896551724138, |
| "grad_norm": 0.5234454954414901, |
| "kl": 0.437255859375, |
| "learning_rate": 1e-06, |
| "loss": 0.003, |
| "reward": 0.7723214477300644, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.7723214477300644, |
| "step": 539 |
| }, |
| { |
| "clip_ratio": 0.00027410148322815076, |
| "epoch": 89.82758620689656, |
| "grad_norm": 0.2959107381794968, |
| "kl": 0.2034912109375, |
| "learning_rate": 1e-06, |
| "loss": 0.0023, |
| "step": 540 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 258.15626525878906, |
| "epoch": 90.13793103448276, |
| "grad_norm": 0.5606663435729826, |
| "kl": 0.2216796875, |
| "learning_rate": 1e-06, |
| "loss": -0.0038, |
| "reward": 0.8839286118745804, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839286118745804, |
| "step": 541 |
| }, |
| { |
| "clip_ratio": 0.0004936900659231469, |
| "epoch": 90.27586206896552, |
| "grad_norm": 0.30058337241291133, |
| "kl": 0.228515625, |
| "learning_rate": 1e-06, |
| "loss": -0.0045, |
| "step": 542 |
| }, |
| { |
| "batch_accuracy": 1.0, |
| "clip_ratio": 0.0, |
| "completion_length": 230.7500114440918, |
| "epoch": 90.41379310344827, |
| "grad_norm": 0.013261529730841225, |
| "kl": 0.1427001953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0001, |
| "reward": 1.0, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 1.0, |
| "step": 543 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 90.55172413793103, |
| "grad_norm": 0.014812055834447134, |
| "kl": 0.146484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0001, |
| "step": 544 |
| }, |
| { |
| "batch_accuracy": 0.8571428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 273.43304443359375, |
| "epoch": 90.6896551724138, |
| "grad_norm": 0.020177764439694507, |
| "kl": 0.1837158203125, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "reward": 0.8571428805589676, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.8571428805589676, |
| "step": 545 |
| }, |
| { |
| "clip_ratio": 0.0, |
| "epoch": 90.82758620689656, |
| "grad_norm": 0.020471063976257625, |
| "kl": 0.184814453125, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "step": 546 |
| }, |
| { |
| "batch_accuracy": 0.8169642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 289.08929443359375, |
| "epoch": 91.13793103448276, |
| "grad_norm": 0.3091335417653999, |
| "kl": 0.270751953125, |
| "learning_rate": 1e-06, |
| "loss": 0.0053, |
| "reward": 0.816964328289032, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.816964328289032, |
| "step": 547 |
| }, |
| { |
| "clip_ratio": 2.406623025308363e-05, |
| "epoch": 91.27586206896552, |
| "grad_norm": 0.1700906815409899, |
| "kl": 0.221435546875, |
| "learning_rate": 1e-06, |
| "loss": 0.0048, |
| "step": 548 |
| }, |
| { |
| "batch_accuracy": 0.9285714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 278.3571586608887, |
| "epoch": 91.41379310344827, |
| "grad_norm": 0.091506341158689, |
| "kl": 0.171630859375, |
| "learning_rate": 1e-06, |
| "loss": 0.0002, |
| "reward": 0.9285714626312256, |
| "reward_std": 0.0, |
| "rewards/unified_reward_func": 0.9285714626312256, |
| "step": 549 |
| }, |
| { |
| "epoch": 91.55172413793103, |
| "grad_norm": 0.014377936909430832, |
| "learning_rate": 1e-06, |
| "loss": 0.0001, |
| "step": 550 |
| }, |
| { |
| "epoch": 91.55172413793103, |
| "eval_batch_accuracy": 0.75, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 273.0262034098307, |
| "eval_kl": 0.3885416666666667, |
| "eval_loss": 0.011643623933196068, |
| "eval_reward": 0.7500000437100728, |
| "eval_reward_std": 0.12268384645382563, |
| "eval_rewards/unified_reward_func": 0.7500000437100728, |
| "eval_runtime": 301.6826, |
| "eval_samples_per_second": 0.331, |
| "eval_steps_per_second": 0.007, |
| "step": 550 |
| }, |
| { |
| "batch_accuracy": 0.8794642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 254.08929824829102, |
| "epoch": 91.6896551724138, |
| "grad_norm": 0.23783649200488552, |
| "kl": 0.18267822265625, |
| "learning_rate": 1e-06, |
| "loss": -0.0033, |
| "reward": 0.8794643431901932, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.8794643431901932, |
| "step": 551 |
| }, |
| { |
| "clip_ratio": 0.00017913367628352717, |
| "epoch": 91.82758620689656, |
| "grad_norm": 0.15644414933150733, |
| "kl": 0.185791015625, |
| "learning_rate": 1e-06, |
| "loss": -0.0036, |
| "step": 552 |
| }, |
| { |
| "batch_accuracy": 0.8303571428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 251.42411422729492, |
| "epoch": 92.13793103448276, |
| "grad_norm": 0.42449283852885444, |
| "kl": 0.187744140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0016, |
| "reward": 0.8303571790456772, |
| "reward_std": 0.04764331132173538, |
| "rewards/unified_reward_func": 0.8303571790456772, |
| "step": 553 |
| }, |
| { |
| "clip_ratio": 0.0004540284280665219, |
| "epoch": 92.27586206896552, |
| "grad_norm": 0.28459956459838937, |
| "kl": 0.18994140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0007, |
| "step": 554 |
| }, |
| { |
| "batch_accuracy": 0.8794642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 269.4062614440918, |
| "epoch": 92.41379310344827, |
| "grad_norm": 0.519460289294838, |
| "kl": 0.41162109375, |
| "learning_rate": 1e-06, |
| "loss": -0.003, |
| "reward": 0.879464328289032, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.879464328289032, |
| "step": 555 |
| }, |
| { |
| "clip_ratio": 0.0007315961556741968, |
| "epoch": 92.55172413793103, |
| "grad_norm": 0.30233640375635246, |
| "kl": 0.505859375, |
| "learning_rate": 1e-06, |
| "loss": -0.0036, |
| "step": 556 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 263.8750114440918, |
| "epoch": 92.6896551724138, |
| "grad_norm": 0.5561022662042301, |
| "kl": 0.641845703125, |
| "learning_rate": 1e-06, |
| "loss": -0.0024, |
| "reward": 0.8839286118745804, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839286118745804, |
| "step": 557 |
| }, |
| { |
| "clip_ratio": 0.0005278162134345621, |
| "epoch": 92.82758620689656, |
| "grad_norm": 0.23851011229091273, |
| "kl": 0.26513671875, |
| "learning_rate": 1e-06, |
| "loss": -0.003, |
| "step": 558 |
| }, |
| { |
| "batch_accuracy": 0.7767857142857142, |
| "clip_ratio": 0.0, |
| "completion_length": 270.8884086608887, |
| "epoch": 93.13793103448276, |
| "grad_norm": 0.4536736642998327, |
| "kl": 0.234375, |
| "learning_rate": 1e-06, |
| "loss": 0.007, |
| "reward": 0.7767857685685158, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.7767857685685158, |
| "step": 559 |
| }, |
| { |
| "clip_ratio": 0.000577432889258489, |
| "epoch": 93.27586206896552, |
| "grad_norm": 0.285893962946527, |
| "kl": 0.244140625, |
| "learning_rate": 1e-06, |
| "loss": 0.0061, |
| "step": 560 |
| }, |
| { |
| "batch_accuracy": 0.9910714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 201.28125762939453, |
| "epoch": 93.41379310344827, |
| "grad_norm": 0.41331881846906104, |
| "kl": 0.231689453125, |
| "learning_rate": 1e-06, |
| "loss": 0.002, |
| "reward": 0.9910714626312256, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9910714626312256, |
| "step": 561 |
| }, |
| { |
| "clip_ratio": 0.0003733747871592641, |
| "epoch": 93.55172413793103, |
| "grad_norm": 0.25037100349015595, |
| "kl": 0.23583984375, |
| "learning_rate": 1e-06, |
| "loss": 0.0013, |
| "step": 562 |
| }, |
| { |
| "batch_accuracy": 0.9017857142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 287.0625114440918, |
| "epoch": 93.6896551724138, |
| "grad_norm": 0.23791341082022063, |
| "kl": 0.342529296875, |
| "learning_rate": 1e-06, |
| "loss": -0.002, |
| "reward": 0.901785746216774, |
| "reward_std": 0.04764330945909023, |
| "rewards/unified_reward_func": 0.901785746216774, |
| "step": 563 |
| }, |
| { |
| "clip_ratio": 0.00034440189483575523, |
| "epoch": 93.82758620689656, |
| "grad_norm": 0.17440834096908142, |
| "kl": 0.340576171875, |
| "learning_rate": 1e-06, |
| "loss": -0.0025, |
| "step": 564 |
| }, |
| { |
| "batch_accuracy": 0.9151785714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 243.61608505249023, |
| "epoch": 94.13793103448276, |
| "grad_norm": 0.6497367763504345, |
| "kl": 0.2451171875, |
| "learning_rate": 1e-06, |
| "loss": 0.0085, |
| "reward": 0.915178582072258, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.915178582072258, |
| "step": 565 |
| }, |
| { |
| "clip_ratio": 0.000318015729135368, |
| "epoch": 94.27586206896552, |
| "grad_norm": 0.7711925543158148, |
| "kl": 0.57373046875, |
| "learning_rate": 1e-06, |
| "loss": 0.0078, |
| "step": 566 |
| }, |
| { |
| "batch_accuracy": 0.9464285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 259.46429443359375, |
| "epoch": 94.41379310344827, |
| "grad_norm": 0.3528493235209404, |
| "kl": 0.35888671875, |
| "learning_rate": 1e-06, |
| "loss": 0.0024, |
| "reward": 0.9464286118745804, |
| "reward_std": 0.03111080639064312, |
| "rewards/unified_reward_func": 0.9464286118745804, |
| "step": 567 |
| }, |
| { |
| "clip_ratio": 0.0003265242121415213, |
| "epoch": 94.55172413793103, |
| "grad_norm": 0.23138814922762713, |
| "kl": 0.36865234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0017, |
| "step": 568 |
| }, |
| { |
| "batch_accuracy": 0.7455357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 253.51340866088867, |
| "epoch": 94.6896551724138, |
| "grad_norm": 0.2436815317174032, |
| "kl": 0.54248046875, |
| "learning_rate": 1e-06, |
| "loss": -0.0038, |
| "reward": 0.7455357611179352, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.7455357611179352, |
| "step": 569 |
| }, |
| { |
| "clip_ratio": 0.0002118285046890378, |
| "epoch": 94.82758620689656, |
| "grad_norm": 0.19930797505222894, |
| "kl": 0.59716796875, |
| "learning_rate": 1e-06, |
| "loss": -0.0041, |
| "step": 570 |
| }, |
| { |
| "batch_accuracy": 0.84375, |
| "clip_ratio": 0.0, |
| "completion_length": 277.7232246398926, |
| "epoch": 95.13793103448276, |
| "grad_norm": 0.6540615887428782, |
| "kl": 0.69287109375, |
| "learning_rate": 1e-06, |
| "loss": -0.0011, |
| "reward": 0.8437500298023224, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.8437500298023224, |
| "step": 571 |
| }, |
| { |
| "clip_ratio": 0.0002021094405790791, |
| "epoch": 95.27586206896552, |
| "grad_norm": 0.32638937252143785, |
| "kl": 0.451904296875, |
| "learning_rate": 1e-06, |
| "loss": -0.002, |
| "step": 572 |
| }, |
| { |
| "batch_accuracy": 0.84375, |
| "clip_ratio": 0.0, |
| "completion_length": 290.8259048461914, |
| "epoch": 95.41379310344827, |
| "grad_norm": 0.32851949325389684, |
| "kl": 0.2333984375, |
| "learning_rate": 1e-06, |
| "loss": -0.02, |
| "reward": 0.8437500447034836, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.8437500447034836, |
| "step": 573 |
| }, |
| { |
| "clip_ratio": 0.00030660181073471904, |
| "epoch": 95.55172413793103, |
| "grad_norm": 0.20845992507237093, |
| "kl": 0.2421875, |
| "learning_rate": 1e-06, |
| "loss": -0.0205, |
| "step": 574 |
| }, |
| { |
| "batch_accuracy": 0.9241071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 217.3259048461914, |
| "epoch": 95.6896551724138, |
| "grad_norm": 0.20862181127671206, |
| "kl": 0.259765625, |
| "learning_rate": 1e-06, |
| "loss": -0.0016, |
| "reward": 0.924107164144516, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.924107164144516, |
| "step": 575 |
| }, |
| { |
| "clip_ratio": 5.621346281259321e-05, |
| "epoch": 95.82758620689656, |
| "grad_norm": 0.15315562157572665, |
| "kl": 0.22265625, |
| "learning_rate": 1e-06, |
| "loss": -0.002, |
| "step": 576 |
| }, |
| { |
| "batch_accuracy": 0.9464285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 255.55358505249023, |
| "epoch": 96.13793103448276, |
| "grad_norm": 0.6579042856320471, |
| "kl": 0.35009765625, |
| "learning_rate": 1e-06, |
| "loss": -0.005, |
| "reward": 0.9464286118745804, |
| "reward_std": 0.04178631864488125, |
| "rewards/unified_reward_func": 0.9464286118745804, |
| "step": 577 |
| }, |
| { |
| "clip_ratio": 0.0008257225781562738, |
| "epoch": 96.27586206896552, |
| "grad_norm": 0.35359186008618476, |
| "kl": 0.25634765625, |
| "learning_rate": 1e-06, |
| "loss": -0.0058, |
| "step": 578 |
| }, |
| { |
| "batch_accuracy": 0.8080357142857143, |
| "clip_ratio": 0.0, |
| "completion_length": 254.8482322692871, |
| "epoch": 96.41379310344827, |
| "grad_norm": 0.41082852610725706, |
| "kl": 0.209716796875, |
| "learning_rate": 1e-06, |
| "loss": 0.0097, |
| "reward": 0.808035746216774, |
| "reward_std": 0.029159409925341606, |
| "rewards/unified_reward_func": 0.808035746216774, |
| "step": 579 |
| }, |
| { |
| "clip_ratio": 0.00018716741033131257, |
| "epoch": 96.55172413793103, |
| "grad_norm": 0.2914419208471443, |
| "kl": 0.218994140625, |
| "learning_rate": 1e-06, |
| "loss": 0.009, |
| "step": 580 |
| }, |
| { |
| "batch_accuracy": 0.9910714285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 255.79018783569336, |
| "epoch": 96.6896551724138, |
| "grad_norm": 0.5785955554778244, |
| "kl": 0.16015625, |
| "learning_rate": 1e-06, |
| "loss": -0.0003, |
| "reward": 0.9910714626312256, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9910714626312256, |
| "step": 581 |
| }, |
| { |
| "clip_ratio": 7.487271068384871e-05, |
| "epoch": 96.82758620689656, |
| "grad_norm": 0.673263897970261, |
| "kl": 0.506591796875, |
| "learning_rate": 1e-06, |
| "loss": -0.0006, |
| "step": 582 |
| }, |
| { |
| "batch_accuracy": 0.9107142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 241.37054443359375, |
| "epoch": 97.13793103448276, |
| "grad_norm": 0.3379560163942377, |
| "kl": 0.175537109375, |
| "learning_rate": 1e-06, |
| "loss": -0.0009, |
| "reward": 0.9107143431901932, |
| "reward_std": 0.03111080639064312, |
| "rewards/unified_reward_func": 0.9107143431901932, |
| "step": 583 |
| }, |
| { |
| "clip_ratio": 0.0002795299296849407, |
| "epoch": 97.27586206896552, |
| "grad_norm": 0.23988831112022546, |
| "kl": 0.18505859375, |
| "learning_rate": 1e-06, |
| "loss": -0.0016, |
| "step": 584 |
| }, |
| { |
| "batch_accuracy": 0.7991071428571428, |
| "clip_ratio": 0.0, |
| "completion_length": 274.97769927978516, |
| "epoch": 97.41379310344827, |
| "grad_norm": 0.504319613213112, |
| "kl": 0.41162109375, |
| "learning_rate": 1e-06, |
| "loss": -0.0038, |
| "reward": 0.7991071939468384, |
| "reward_std": 0.05441322363913059, |
| "rewards/unified_reward_func": 0.7991071939468384, |
| "step": 585 |
| }, |
| { |
| "clip_ratio": 0.0004081932838744251, |
| "epoch": 97.55172413793103, |
| "grad_norm": 0.40195943010535184, |
| "kl": 0.310302734375, |
| "learning_rate": 1e-06, |
| "loss": -0.0046, |
| "step": 586 |
| }, |
| { |
| "batch_accuracy": 0.8839285714285714, |
| "clip_ratio": 0.0, |
| "completion_length": 304.45983505249023, |
| "epoch": 97.6896551724138, |
| "grad_norm": 153.18400076986597, |
| "kl": 27.049072265625, |
| "learning_rate": 1e-06, |
| "loss": 0.0295, |
| "reward": 0.8839285969734192, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.8839285969734192, |
| "step": 587 |
| }, |
| { |
| "clip_ratio": 0.00023562640853924677, |
| "epoch": 97.82758620689656, |
| "grad_norm": 0.2820135906671078, |
| "kl": 0.3720703125, |
| "learning_rate": 1e-06, |
| "loss": 0.0028, |
| "step": 588 |
| }, |
| { |
| "batch_accuracy": 0.9508928571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 296.3303756713867, |
| "epoch": 98.13793103448276, |
| "grad_norm": 0.3651174815481574, |
| "kl": 0.28564453125, |
| "learning_rate": 1e-06, |
| "loss": -0.0003, |
| "reward": 0.9508928954601288, |
| "reward_std": 0.03788072057068348, |
| "rewards/unified_reward_func": 0.9508928954601288, |
| "step": 589 |
| }, |
| { |
| "clip_ratio": 0.00031237147777574137, |
| "epoch": 98.27586206896552, |
| "grad_norm": 0.2478077249618373, |
| "kl": 0.2294921875, |
| "learning_rate": 1e-06, |
| "loss": -0.0009, |
| "step": 590 |
| }, |
| { |
| "batch_accuracy": 0.8348214285714286, |
| "clip_ratio": 0.0, |
| "completion_length": 271.37500762939453, |
| "epoch": 98.41379310344827, |
| "grad_norm": 0.5379304215920322, |
| "kl": 0.42138671875, |
| "learning_rate": 1e-06, |
| "loss": -0.0008, |
| "reward": 0.8348214626312256, |
| "reward_std": 0.05441322363913059, |
| "rewards/unified_reward_func": 0.8348214626312256, |
| "step": 591 |
| }, |
| { |
| "clip_ratio": 0.0003275511880929116, |
| "epoch": 98.55172413793103, |
| "grad_norm": 11.095997641379615, |
| "kl": 0.21484375, |
| "learning_rate": 1e-06, |
| "loss": 0.0037, |
| "step": 592 |
| }, |
| { |
| "batch_accuracy": 0.90625, |
| "clip_ratio": 0.0, |
| "completion_length": 258.28126525878906, |
| "epoch": 98.6896551724138, |
| "grad_norm": 0.5801422636522526, |
| "kl": 0.224365234375, |
| "learning_rate": 1e-06, |
| "loss": 0.0021, |
| "reward": 0.9062500447034836, |
| "reward_std": 0.06313453242182732, |
| "rewards/unified_reward_func": 0.9062500447034836, |
| "step": 593 |
| }, |
| { |
| "clip_ratio": 0.0003545371364452876, |
| "epoch": 98.82758620689656, |
| "grad_norm": 0.6215554312278989, |
| "kl": 0.388916015625, |
| "learning_rate": 1e-06, |
| "loss": 0.0014, |
| "step": 594 |
| }, |
| { |
| "batch_accuracy": 0.8169642857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 276.9107246398926, |
| "epoch": 99.13793103448276, |
| "grad_norm": 6.009087004919584, |
| "kl": 1.102294921875, |
| "learning_rate": 1e-06, |
| "loss": -0.0008, |
| "reward": 0.8169643133878708, |
| "reward_std": 0.012626906856894493, |
| "rewards/unified_reward_func": 0.8169643133878708, |
| "step": 595 |
| }, |
| { |
| "clip_ratio": 9.897070412989706e-05, |
| "epoch": 99.27586206896552, |
| "grad_norm": 0.14876236532911624, |
| "kl": 0.203125, |
| "learning_rate": 1e-06, |
| "loss": -0.0017, |
| "step": 596 |
| }, |
| { |
| "batch_accuracy": 0.9196428571428571, |
| "clip_ratio": 0.0, |
| "completion_length": 260.433048248291, |
| "epoch": 99.41379310344827, |
| "grad_norm": 0.25737299250337803, |
| "kl": 0.283447265625, |
| "learning_rate": 1e-06, |
| "loss": -0.0059, |
| "reward": 0.9196428954601288, |
| "reward_std": 0.025253813713788986, |
| "rewards/unified_reward_func": 0.9196428954601288, |
| "step": 597 |
| }, |
| { |
| "clip_ratio": 0.00015474966494366527, |
| "epoch": 99.55172413793103, |
| "grad_norm": 0.17163873622166492, |
| "kl": 0.310546875, |
| "learning_rate": 1e-06, |
| "loss": -0.0063, |
| "step": 598 |
| }, |
| { |
| "batch_accuracy": 0.9107142857142857, |
| "clip_ratio": 0.0, |
| "completion_length": 285.45537185668945, |
| "epoch": 99.6896551724138, |
| "grad_norm": 0.4237895842134686, |
| "kl": 0.296630859375, |
| "learning_rate": 1e-06, |
| "loss": 0.003, |
| "reward": 0.9107142984867096, |
| "reward_std": 0.04178631864488125, |
| "rewards/unified_reward_func": 0.9107142984867096, |
| "step": 599 |
| }, |
| { |
| "epoch": 99.82758620689656, |
| "grad_norm": 0.33686343212751796, |
| "learning_rate": 1e-06, |
| "loss": 0.0023, |
| "step": 600 |
| }, |
| { |
| "epoch": 99.82758620689656, |
| "eval_batch_accuracy": 0.75, |
| "eval_clip_ratio": 0.0, |
| "eval_completion_length": 292.19373779296876, |
| "eval_kl": 0.08203125, |
| "eval_loss": -0.002871450036764145, |
| "eval_reward": 0.7500000357627868, |
| "eval_reward_std": 0.11066367477178574, |
| "eval_rewards/unified_reward_func": 0.7500000357627868, |
| "eval_runtime": 317.6626, |
| "eval_samples_per_second": 0.315, |
| "eval_steps_per_second": 0.006, |
| "step": 600 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 700, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 100, |
| "save_steps": 100, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 8, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|