diff --git a/healed/grid_general_fairness/glean_keep50_s1224_long500.console.log b/healed/grid_general_fairness/glean_keep50_s1224_long500.console.log new file mode 100644 index 0000000000000000000000000000000000000000..cf657e74358cb8f5a53e695e7736d8b60ebe8388 --- /dev/null +++ b/healed/grid_general_fairness/glean_keep50_s1224_long500.console.log @@ -0,0 +1,602 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/wandb/run-20260719_142238-77wxfx9w +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run glean_keep50_s1224_long500 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-general-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-general-grid/runs/77wxfx9w + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0150 +{"step": 151, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3593016570910501, "tokens": 120000, "cumulative_loss_tokens": 18120000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 317.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.6, "frames": {"chat": 378}, "mem_gb": 15.78} +{"step": 152, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4341406711792573, "tokens": 120000, "cumulative_loss_tokens": 18240000, "grad_norm": 0.6953125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 326.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.8, "frames": {"chat": 368}, "mem_gb": 15.88} +{"step": 153, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18279959440904203, "tokens": 120000, "cumulative_loss_tokens": 18360000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.898, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 256}, "mem_gb": 15.99} +{"step": 154, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20074472173604493, "tokens": 120000, "cumulative_loss_tokens": 18480000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.93, "comp_len": 381.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 315}, "mem_gb": 15.93} +{"step": 155, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18954649618460487, "tokens": 120000, "cumulative_loss_tokens": 18600000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.974, "comp_len": 384.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 312}, "mem_gb": 15.86} +{"step": 156, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2893579348502991, "tokens": 120000, "cumulative_loss_tokens": 18720000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.97, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 330}, "mem_gb": 16.04} +{"step": 157, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25195731372358277, "tokens": 120000, "cumulative_loss_tokens": 18840000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.954, "comp_len": 367.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.6, "frames": {"chat": 327}, "mem_gb": 16.05} +{"step": 158, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3365387867776056, "tokens": 120000, "cumulative_loss_tokens": 18960000, "grad_norm": 0.6171875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 341.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.2, "frames": {"chat": 351}, "mem_gb": 15.7} +{"step": 159, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19302456602317591, "tokens": 120000, "cumulative_loss_tokens": 19080000, "grad_norm": 0.439453125, "lr": 3e-05, "finish_rate": 0.973, "comp_len": 357.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 336}, "mem_gb": 15.76} +{"step": 160, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3532588432255822, "tokens": 120000, "cumulative_loss_tokens": 19200000, "grad_norm": 0.7109375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 332.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.6, "frames": {"chat": 361}, "mem_gb": 15.73} +[eval step 160] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1 & ' +{"step": 161, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.37139201210066675, "tokens": 120000, "cumulative_loss_tokens": 19320000, "grad_norm": 0.6171875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 328.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.8, "frames": {"chat": 365}, "mem_gb": 15.68} +{"step": 162, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3304392444904273, "tokens": 120000, "cumulative_loss_tokens": 19440000, "grad_norm": 0.5859375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 364.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 329}, "mem_gb": 15.73} +{"step": 163, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2241786515767686, "tokens": 120000, "cumulative_loss_tokens": 19560000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.931, "comp_len": 416.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 288}, "mem_gb": 15.94} +{"step": 164, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16131471410921464, "tokens": 120000, "cumulative_loss_tokens": 19680000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 451.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 266}, "mem_gb": 16.09} +{"step": 165, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2982499245635544, "tokens": 120000, "cumulative_loss_tokens": 19800000, "grad_norm": 0.5546875, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 372.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 322}, "mem_gb": 16.01} +{"step": 166, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3000292730904805, "tokens": 120000, "cumulative_loss_tokens": 19920000, "grad_norm": 0.5625, "lr": 3e-05, "finish_rate": 0.989, "comp_len": 334.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.4, "frames": {"chat": 359}, "mem_gb": 15.71} +{"step": 167, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2509844661024399, "tokens": 120000, "cumulative_loss_tokens": 20040000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 385.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 311}, "mem_gb": 15.83} +{"step": 168, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22655330060627313, "tokens": 120000, "cumulative_loss_tokens": 20160000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.95, "comp_len": 396.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 303}, "mem_gb": 15.88} +{"step": 169, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3782985839564974, "tokens": 120000, "cumulative_loss_tokens": 20280000, "grad_norm": 0.625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 351.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.1, "frames": {"chat": 341}, "mem_gb": 15.89} +{"step": 170, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08561049039121717, "tokens": 120000, "cumulative_loss_tokens": 20400000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.0, "frames": {"chat": 229}, "mem_gb": 15.95} +[eval step 170] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -' +{"step": 171, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23171288729174994, "tokens": 120000, "cumulative_loss_tokens": 20520000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.964, "comp_len": 389.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 308}, "mem_gb": 16.11} +{"step": 172, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32432291762831933, "tokens": 120000, "cumulative_loss_tokens": 20640000, "grad_norm": 0.57421875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 331.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.2, "frames": {"chat": 362}, "mem_gb": 15.61} +{"step": 173, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3459035145717673, "tokens": 120000, "cumulative_loss_tokens": 20760000, "grad_norm": 0.60546875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 315.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.8, "frames": {"chat": 380}, "mem_gb": 15.85} +{"step": 174, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.31912819158503164, "tokens": 120000, "cumulative_loss_tokens": 20880000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 336.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.2, "frames": {"chat": 357}, "mem_gb": 15.63} +{"step": 175, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1714131233768848, "tokens": 120000, "cumulative_loss_tokens": 21000000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 254}, "mem_gb": 16.11} +{"step": 176, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13373881205534563, "tokens": 120000, "cumulative_loss_tokens": 21120000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.874, "comp_len": 459.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 261}, "mem_gb": 16.08} +{"step": 177, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2999917506314814, "tokens": 120000, "cumulative_loss_tokens": 21240000, "grad_norm": 0.57421875, "lr": 3e-05, "finish_rate": 0.982, "comp_len": 431.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 278}, "mem_gb": 15.87} +{"step": 178, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3210427028543161, "tokens": 120000, "cumulative_loss_tokens": 21360000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 360.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.3, "frames": {"chat": 333}, "mem_gb": 16.0} +{"step": 179, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15903221555058844, "tokens": 120000, "cumulative_loss_tokens": 21480000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 250}, "mem_gb": 15.97} +{"step": 180, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3951781343491127, "tokens": 120000, "cumulative_loss_tokens": 21600000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 336.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.9, "frames": {"chat": 357}, "mem_gb": 15.75} +[eval step 180] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -' +{"step": 181, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4099205974268261, "tokens": 120000, "cumulative_loss_tokens": 21720000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.7, "frames": {"chat": 330}, "mem_gb": 15.71} +{"step": 182, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3410182152831306, "tokens": 120000, "cumulative_loss_tokens": 21840000, "grad_norm": 0.60546875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 341.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.8, "frames": {"chat": 351}, "mem_gb": 15.75} +{"step": 183, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2245930080806526, "tokens": 120000, "cumulative_loss_tokens": 21960000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 413.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 290}, "mem_gb": 15.88} +{"step": 184, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2883827197744977, "tokens": 120000, "cumulative_loss_tokens": 22080000, "grad_norm": 0.55078125, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 372.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.0, "frames": {"chat": 322}, "mem_gb": 15.83} +{"step": 185, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1481790694156351, "tokens": 120000, "cumulative_loss_tokens": 22200000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 451.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 266}, "mem_gb": 16.06} +{"step": 186, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2764926481295998, "tokens": 120000, "cumulative_loss_tokens": 22320000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.4, "frames": {"chat": 331}, "mem_gb": 15.91} +{"step": 187, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.25685157680145154, "tokens": 120000, "cumulative_loss_tokens": 22440000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.8, "frames": {"chat": 342}, "mem_gb": 15.67} +{"step": 188, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.29339611131663745, "tokens": 120000, "cumulative_loss_tokens": 22560000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.4, "frames": {"chat": 358}, "mem_gb": 15.71} +{"step": 189, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2561505928792991, "tokens": 120000, "cumulative_loss_tokens": 22680000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 348.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.3, "frames": {"chat": 344}, "mem_gb": 15.79} +{"step": 190, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15278181020182868, "tokens": 120000, "cumulative_loss_tokens": 22800000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 449.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 267}, "mem_gb": 16.11} +[eval step 190] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1 &' +{"step": 191, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.057855343678841986, "tokens": 120000, "cumulative_loss_tokens": 22920000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 248}, "mem_gb": 15.94} +{"step": 192, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23286335130035876, "tokens": 120000, "cumulative_loss_tokens": 23040000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.979, "comp_len": 364.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 329}, "mem_gb": 15.93} +{"step": 193, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24648634702761968, "tokens": 120000, "cumulative_loss_tokens": 23160000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 354.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.0, "frames": {"chat": 339}, "mem_gb": 15.89} +{"step": 194, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.28432376994520114, "tokens": 120000, "cumulative_loss_tokens": 23280000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.3, "frames": {"chat": 354}, "mem_gb": 15.89} +{"step": 195, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15807348166286636, "tokens": 120000, "cumulative_loss_tokens": 23400000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.943, "comp_len": 430.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 279}, "mem_gb": 16.02} +{"step": 196, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19517099342911193, "tokens": 120000, "cumulative_loss_tokens": 23520000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 318}, "mem_gb": 15.95} +{"step": 197, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12892133611446868, "tokens": 120000, "cumulative_loss_tokens": 23640000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.875, "comp_len": 452.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 265}, "mem_gb": 16.04} +{"step": 198, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12192254397664219, "tokens": 120000, "cumulative_loss_tokens": 23760000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.85, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 254}, "mem_gb": 15.98} +{"step": 199, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1897617371759222, "tokens": 120000, "cumulative_loss_tokens": 23880000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.962, "comp_len": 381.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 315}, "mem_gb": 16.01} +{"step": 200, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2745921495852992, "tokens": 120000, "cumulative_loss_tokens": 24000000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 300}, "mem_gb": 15.81} +[eval step 200] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly independent rows or col' +{"step": 201, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16728228757098937, "tokens": 120000, "cumulative_loss_tokens": 24120000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.924, "comp_len": 434.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 276}, "mem_gb": 16.01} +{"step": 202, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2832062610987574, "tokens": 120000, "cumulative_loss_tokens": 24240000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 337.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.3, "frames": {"chat": 356}, "mem_gb": 15.95} +{"step": 203, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.258948408873938, "tokens": 120000, "cumulative_loss_tokens": 24360000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 394.7, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 304}, "mem_gb": 15.77} +{"step": 204, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2657215355453237, "tokens": 120000, "cumulative_loss_tokens": 24480000, "grad_norm": 0.4921875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 320.0, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 57.9, "frames": {"chat": 375}, "mem_gb": 15.68} +{"step": 205, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12172198731293901, "tokens": 120000, "cumulative_loss_tokens": 24600000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.939, "comp_len": 408.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 294}, "mem_gb": 16.04} +{"step": 206, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2380220364671511, "tokens": 120000, "cumulative_loss_tokens": 24720000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 401.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 299}, "mem_gb": 16.05} +{"step": 207, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1718156932564142, "tokens": 120000, "cumulative_loss_tokens": 24840000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 393.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 305}, "mem_gb": 15.83} +{"step": 208, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10488310244778792, "tokens": 120000, "cumulative_loss_tokens": 24960000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.931, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 262}, "mem_gb": 15.88} +{"step": 209, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.153289779908179, "tokens": 120000, "cumulative_loss_tokens": 25080000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.95, "comp_len": 396.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 303}, "mem_gb": 15.94} +{"step": 210, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.26352043294856947, "tokens": 120000, "cumulative_loss_tokens": 25200000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 315.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.1, "frames": {"chat": 380}, "mem_gb": 15.82} +[eval step 210] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\\[ A = \\begin{bmatrix}\n12 & " +{"step": 211, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11769829866392538, "tokens": 120000, "cumulative_loss_tokens": 25320000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.95, "comp_len": 425.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 282}, "mem_gb": 15.78} +{"step": 212, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24135005666119977, "tokens": 120000, "cumulative_loss_tokens": 25440000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 331.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.8, "frames": {"chat": 362}, "mem_gb": 15.79} +{"step": 213, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23439129272607775, "tokens": 120000, "cumulative_loss_tokens": 25560000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.967, "comp_len": 390.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.2, "frames": {"chat": 307}, "mem_gb": 16.04} +{"step": 214, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15147519016768785, "tokens": 120000, "cumulative_loss_tokens": 25680000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 425.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.5, "frames": {"chat": 282}, "mem_gb": 15.9} +{"step": 215, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24968825165581268, "tokens": 120000, "cumulative_loss_tokens": 25800000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.979, "comp_len": 356.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.4, "frames": {"chat": 337}, "mem_gb": 15.93} +{"step": 216, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14156633071006897, "tokens": 120000, "cumulative_loss_tokens": 25920000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.891, "comp_len": 436.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 275}, "mem_gb": 15.99} +{"step": 217, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2628871244806175, "tokens": 120000, "cumulative_loss_tokens": 26040000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 318}, "mem_gb": 15.8} +{"step": 218, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3056329848428412, "tokens": 120000, "cumulative_loss_tokens": 26160000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 311.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.8, "frames": {"chat": 385}, "mem_gb": 15.83} +{"step": 219, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3026575644499622, "tokens": 120000, "cumulative_loss_tokens": 26280000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 379.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 316}, "mem_gb": 15.78} +{"step": 220, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10592310933292222, "tokens": 120000, "cumulative_loss_tokens": 26400000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 451.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 266}, "mem_gb": 16.07} +[eval step 220] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given 4x4 matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & " +{"step": 221, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.268542345561025, "tokens": 120000, "cumulative_loss_tokens": 26520000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 323.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.2, "frames": {"chat": 371}, "mem_gb": 15.8} +{"step": 222, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13127939274306408, "tokens": 120000, "cumulative_loss_tokens": 26640000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 262}, "mem_gb": 16.01} +{"step": 223, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.28847831037867194, "tokens": 120000, "cumulative_loss_tokens": 26760000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 344.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.9, "frames": {"chat": 348}, "mem_gb": 15.83} +{"step": 224, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04362238221272516, "tokens": 120000, "cumulative_loss_tokens": 26880000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.7, "frames": {"chat": 222}, "mem_gb": 16.07} +{"step": 225, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24061391098311483, "tokens": 120000, "cumulative_loss_tokens": 27000000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.99, "comp_len": 294.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.3, "frames": {"chat": 408}, "mem_gb": 15.63} +{"step": 226, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23265060038057467, "tokens": 120000, "cumulative_loss_tokens": 27120000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 359.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 334}, "mem_gb": 15.7} +{"step": 227, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11811345087826873, "tokens": 120000, "cumulative_loss_tokens": 27240000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.968, "comp_len": 384.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 312}, "mem_gb": 15.75} +{"step": 228, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.237524250674434, "tokens": 120000, "cumulative_loss_tokens": 27360000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 340.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.8, "frames": {"chat": 352}, "mem_gb": 15.73} +{"step": 229, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20335884417813893, "tokens": 120000, "cumulative_loss_tokens": 27480000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 323.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.3, "frames": {"chat": 371}, "mem_gb": 15.54} +{"step": 230, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1350035905787566, "tokens": 120000, "cumulative_loss_tokens": 27600000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.898, "comp_len": 422.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 284}, "mem_gb": 15.99} +[eval step 230] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\\[ A = \\begin{bmatrix}\n12 & " +{"step": 231, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11093337530991994, "tokens": 120000, "cumulative_loss_tokens": 27720000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 418.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 287}, "mem_gb": 16.06} +{"step": 232, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09499362250724808, "tokens": 120000, "cumulative_loss_tokens": 27840000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.916, "comp_len": 421.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 285}, "mem_gb": 15.95} +{"step": 233, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.058406724956794644, "tokens": 120000, "cumulative_loss_tokens": 27960000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.873, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.3, "frames": {"chat": 252}, "mem_gb": 16.04} +{"step": 234, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24381569011175694, "tokens": 120000, "cumulative_loss_tokens": 28080000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 333.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.5, "frames": {"chat": 360}, "mem_gb": 15.83} +{"step": 235, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.25821272418865315, "tokens": 120000, "cumulative_loss_tokens": 28200000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 334.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.3, "frames": {"chat": 359}, "mem_gb": 15.82} +{"step": 236, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2568633184568258, "tokens": 120000, "cumulative_loss_tokens": 28320000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 1.0, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.9, "frames": {"chat": 363}, "mem_gb": 15.58} +{"step": 237, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22184100167021775, "tokens": 120000, "cumulative_loss_tokens": 28440000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 332.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.7, "frames": {"chat": 361}, "mem_gb": 15.67} +{"step": 238, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13508064506764833, "tokens": 120000, "cumulative_loss_tokens": 28560000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.944, "comp_len": 419.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 286}, "mem_gb": 15.86} +{"step": 239, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2511199402528815, "tokens": 120000, "cumulative_loss_tokens": 28680000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 327.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.6, "frames": {"chat": 367}, "mem_gb": 15.71} +{"step": 240, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2663104225285972, "tokens": 120000, "cumulative_loss_tokens": 28800000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 340.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.3, "frames": {"chat": 352}, "mem_gb": 15.74} +[eval step 240] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix \\( A \\) as:\n\\[ A = \\begin{pmatrix}\n12" +{"step": 241, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15247889521208902, "tokens": 120000, "cumulative_loss_tokens": 28920000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.962, "comp_len": 382.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 314}, "mem_gb": 15.69} +{"step": 242, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23093722466112426, "tokens": 120000, "cumulative_loss_tokens": 29040000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.995, "comp_len": 322.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.1, "frames": {"chat": 372}, "mem_gb": 15.66} +{"step": 243, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20221049303791175, "tokens": 120000, "cumulative_loss_tokens": 29160000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 359.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.4, "frames": {"chat": 334}, "mem_gb": 15.81} +{"step": 244, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2734870941676044, "tokens": 120000, "cumulative_loss_tokens": 29280000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 326.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.8, "frames": {"chat": 368}, "mem_gb": 15.75} +{"step": 245, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20306090018311515, "tokens": 120000, "cumulative_loss_tokens": 29400000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.994, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.7, "frames": {"chat": 354}, "mem_gb": 15.9} +{"step": 246, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11651120234581952, "tokens": 120000, "cumulative_loss_tokens": 29520000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.973, "comp_len": 409.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 293}, "mem_gb": 15.79} +{"step": 247, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06376467944063867, "tokens": 120000, "cumulative_loss_tokens": 29640000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 231}, "mem_gb": 15.95} +{"step": 248, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18447236588636104, "tokens": 120000, "cumulative_loss_tokens": 29760000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.964, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 302}, "mem_gb": 15.92} +{"step": 249, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1227916136342256, "tokens": 120000, "cumulative_loss_tokens": 29880000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 456.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 263}, "mem_gb": 16.06} +{"step": 250, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13438162850821392, "tokens": 120000, "cumulative_loss_tokens": 30000000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 252}, "mem_gb": 15.99} +[eval step 250] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix \\( A \\) as:\n\\[ A = \\begin{bmatrix}\n12" +{"step": 251, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19821062134134892, "tokens": 120000, "cumulative_loss_tokens": 30120000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.968, "comp_len": 348.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.5, "frames": {"chat": 344}, "mem_gb": 16.02} +{"step": 252, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18651708286859406, "tokens": 120000, "cumulative_loss_tokens": 30240000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.981, "comp_len": 383.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.9, "frames": {"chat": 313}, "mem_gb": 15.95} +{"step": 253, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2265252312041043, "tokens": 120000, "cumulative_loss_tokens": 30360000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.992, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.2, "frames": {"chat": 358}, "mem_gb": 15.65} +{"step": 254, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10849972841935232, "tokens": 120000, "cumulative_loss_tokens": 30480000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.951, "comp_len": 449.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.9, "frames": {"chat": 267}, "mem_gb": 15.87} +{"step": 255, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22024506733524613, "tokens": 120000, "cumulative_loss_tokens": 30600000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 300}, "mem_gb": 15.74} +{"step": 256, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18654347909581848, "tokens": 120000, "cumulative_loss_tokens": 30720000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.994, "comp_len": 344.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.5, "frames": {"chat": 348}, "mem_gb": 15.68} +{"step": 257, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23171534074228256, "tokens": 120000, "cumulative_loss_tokens": 30840000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 364.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.3, "frames": {"chat": 329}, "mem_gb": 15.77} +{"step": 258, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1187580151544263, "tokens": 120000, "cumulative_loss_tokens": 30960000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.948, "comp_len": 446.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 269}, "mem_gb": 15.93} +{"step": 259, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24210104198073967, "tokens": 120000, "cumulative_loss_tokens": 31080000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.6, "frames": {"chat": 358}, "mem_gb": 15.6} +{"step": 260, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1402670572983877, "tokens": 120000, "cumulative_loss_tokens": 31200000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.96, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 302}, "mem_gb": 15.86} +[eval step 260] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -' +{"step": 261, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24428478523382607, "tokens": 120000, "cumulative_loss_tokens": 31320000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 315.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.2, "frames": {"chat": 381}, "mem_gb": 15.87} +{"step": 262, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2672723070027617, "tokens": 120000, "cumulative_loss_tokens": 31440000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 316.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.2, "frames": {"chat": 379}, "mem_gb": 15.82} +{"step": 263, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2640114396764586, "tokens": 120000, "cumulative_loss_tokens": 31560000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 341.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.6, "frames": {"chat": 351}, "mem_gb": 15.64} +{"step": 264, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21936256886639943, "tokens": 120000, "cumulative_loss_tokens": 31680000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 330}, "mem_gb": 15.84} +{"step": 265, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22622612311318516, "tokens": 120000, "cumulative_loss_tokens": 31800000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.991, "comp_len": 379.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.3, "frames": {"chat": 316}, "mem_gb": 15.87} +{"step": 266, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14763944800250853, "tokens": 120000, "cumulative_loss_tokens": 31920000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.891, "comp_len": 434.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 276}, "mem_gb": 16.05} +{"step": 267, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12709228399413017, "tokens": 120000, "cumulative_loss_tokens": 32040000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 240}, "mem_gb": 15.99} +{"step": 268, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13622348031157938, "tokens": 120000, "cumulative_loss_tokens": 32160000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.948, "comp_len": 416.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.5, "frames": {"chat": 288}, "mem_gb": 15.9} +{"step": 269, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12852995432841902, "tokens": 120000, "cumulative_loss_tokens": 32280000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.937, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 271}, "mem_gb": 15.89} +{"step": 270, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20651311299906422, "tokens": 120000, "cumulative_loss_tokens": 32400000, "grad_norm": 0.474609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 336.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.0, "frames": {"chat": 357}, "mem_gb": 15.84} +[eval step 270] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly indepe' +{"step": 271, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.27526303681734327, "tokens": 120000, "cumulative_loss_tokens": 32520000, "grad_norm": 0.52734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.1, "frames": {"chat": 354}, "mem_gb": 15.81} +{"step": 272, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2642003723702083, "tokens": 120000, "cumulative_loss_tokens": 32640000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.927, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 302}, "mem_gb": 16.05} +{"step": 273, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19161327272173947, "tokens": 120000, "cumulative_loss_tokens": 32760000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 404.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 297}, "mem_gb": 16.04} +{"step": 274, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05750741915637627, "tokens": 120000, "cumulative_loss_tokens": 32880000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.8, "frames": {"chat": 219}, "mem_gb": 16.06} +{"step": 275, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2618174074056248, "tokens": 120000, "cumulative_loss_tokens": 33000000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.99, "comp_len": 392.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 306}, "mem_gb": 15.76} +{"step": 276, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.259288263825234, "tokens": 120000, "cumulative_loss_tokens": 33120000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 349.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 343}, "mem_gb": 15.7} +{"step": 277, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2392211010082004, "tokens": 120000, "cumulative_loss_tokens": 33240000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 351.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.1, "frames": {"chat": 341}, "mem_gb": 15.81} +{"step": 278, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14987188792719194, "tokens": 120000, "cumulative_loss_tokens": 33360000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 318}, "mem_gb": 15.76} +{"step": 279, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2518374318243315, "tokens": 120000, "cumulative_loss_tokens": 33480000, "grad_norm": 0.490234375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 346.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.2, "frames": {"chat": 346}, "mem_gb": 15.65} +{"step": 280, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1228407441490485, "tokens": 120000, "cumulative_loss_tokens": 33600000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 433.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 277}, "mem_gb": 16.13} +[eval step 280] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n12 &" +{"step": 281, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11820770578325415, "tokens": 120000, "cumulative_loss_tokens": 33720000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.956, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 270}, "mem_gb": 15.86} +{"step": 282, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15812498243773976, "tokens": 120000, "cumulative_loss_tokens": 33840000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 257}, "mem_gb": 16.04} +{"step": 283, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2444414727439483, "tokens": 120000, "cumulative_loss_tokens": 33960000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.3, "frames": {"chat": 363}, "mem_gb": 15.65} +{"step": 284, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.25830430121341874, "tokens": 120000, "cumulative_loss_tokens": 34080000, "grad_norm": 0.474609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 337.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.4, "frames": {"chat": 356}, "mem_gb": 15.79} +{"step": 285, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22498066962643837, "tokens": 120000, "cumulative_loss_tokens": 34200000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 371.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 323}, "mem_gb": 15.82} +{"step": 286, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1076765178909991, "tokens": 120000, "cumulative_loss_tokens": 34320000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.8, "frames": {"chat": 243}, "mem_gb": 16.07} +{"step": 287, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22851032345732675, "tokens": 120000, "cumulative_loss_tokens": 34440000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 390.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 307}, "mem_gb": 15.75} +{"step": 288, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21784643366625533, "tokens": 120000, "cumulative_loss_tokens": 34560000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 346.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 346}, "mem_gb": 15.99} +{"step": 289, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13869814217598178, "tokens": 120000, "cumulative_loss_tokens": 34680000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.905, "comp_len": 439.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 273}, "mem_gb": 15.99} +{"step": 290, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14725039266501552, "tokens": 120000, "cumulative_loss_tokens": 34800000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.957, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 302}, "mem_gb": 15.92} +[eval step 290] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns. In this 4x4 matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1 & -10 \\\\\n0 & 1' +{"step": 291, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2211726140391392, "tokens": 120000, "cumulative_loss_tokens": 34920000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.97, "comp_len": 356.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 337}, "mem_gb": 16.05} +{"step": 292, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16150297817612688, "tokens": 120000, "cumulative_loss_tokens": 35040000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 428.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 280}, "mem_gb": 15.93} +{"step": 293, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2763836035844094, "tokens": 120000, "cumulative_loss_tokens": 35160000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 331}, "mem_gb": 15.83} +{"step": 294, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16249694327288308, "tokens": 120000, "cumulative_loss_tokens": 35280000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 382.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 314}, "mem_gb": 15.93} +{"step": 295, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.28225345040137567, "tokens": 120000, "cumulative_loss_tokens": 35400000, "grad_norm": 0.490234375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 408.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 294}, "mem_gb": 15.86} +{"step": 296, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12327918875319883, "tokens": 120000, "cumulative_loss_tokens": 35520000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.894, "comp_len": 438.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 274}, "mem_gb": 15.99} +{"step": 297, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21619978666777412, "tokens": 120000, "cumulative_loss_tokens": 35640000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.995, "comp_len": 328.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.1, "frames": {"chat": 365}, "mem_gb": 15.71} +{"step": 298, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13386822491961065, "tokens": 120000, "cumulative_loss_tokens": 35760000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.972, "comp_len": 371.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 323}, "mem_gb": 15.77} +{"step": 299, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0922729125728365, "tokens": 120000, "cumulative_loss_tokens": 35880000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.829, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.8, "frames": {"chat": 234}, "mem_gb": 16.06} +{"step": 300, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12687125249713038, "tokens": 120000, "cumulative_loss_tokens": 36000000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.891, "comp_len": 449.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 267}, "mem_gb": 16.06} +[eval step 300] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given 4x4 matrix \\( A \\) as follows:\n\\[ A = \\begin{pm" +checkpoint snapshot queued -> outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0300 +{"step": 301, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14853974011619575, "tokens": 120000, "cumulative_loss_tokens": 36120000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.961, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 304}, "mem_gb": 15.82} +{"step": 302, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23704376627886667, "tokens": 120000, "cumulative_loss_tokens": 36240000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 321.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.8, "frames": {"chat": 373}, "mem_gb": 15.75} +{"step": 303, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13807657204434898, "tokens": 120000, "cumulative_loss_tokens": 36360000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 300}, "mem_gb": 16.03} +{"step": 304, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24931503556435927, "tokens": 120000, "cumulative_loss_tokens": 36480000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.982, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.3, "frames": {"chat": 331}, "mem_gb": 15.87} +{"step": 305, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10890925987434263, "tokens": 120000, "cumulative_loss_tokens": 36600000, "grad_norm": 0.71484375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 259}, "mem_gb": 16.01} +{"step": 306, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21857231575131106, "tokens": 120000, "cumulative_loss_tokens": 36720000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.6, "frames": {"chat": 358}, "mem_gb": 15.63} +{"step": 307, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2661008476445141, "tokens": 120000, "cumulative_loss_tokens": 36840000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.0, "frames": {"chat": 354}, "mem_gb": 15.88} +{"step": 308, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12416610778168154, "tokens": 120000, "cumulative_loss_tokens": 36960000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.939, "comp_len": 430.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 279}, "mem_gb": 15.95} +{"step": 309, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2604289408598406, "tokens": 120000, "cumulative_loss_tokens": 37080000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.4, "frames": {"chat": 354}, "mem_gb": 15.7} +{"step": 310, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2566351361444375, "tokens": 120000, "cumulative_loss_tokens": 37200000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 359.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.7, "frames": {"chat": 334}, "mem_gb": 15.9} +[eval step 310] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given 4x4 matrix as \\( A \\):\n\n\\[ A = \\begin{bmatri" +{"step": 311, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.258785496380801, "tokens": 120000, "cumulative_loss_tokens": 37320000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 367.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.8, "frames": {"chat": 327}, "mem_gb": 15.77} +{"step": 312, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2259171037707245, "tokens": 120000, "cumulative_loss_tokens": 37440000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 328.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.8, "frames": {"chat": 365}, "mem_gb": 15.87} +{"step": 313, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1300395100544362, "tokens": 120000, "cumulative_loss_tokens": 37560000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.869, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 252}, "mem_gb": 16.08} +{"step": 314, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1075184213388556, "tokens": 120000, "cumulative_loss_tokens": 37680000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.7, "frames": {"chat": 256}, "mem_gb": 16.04} +{"step": 315, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1376660783306385, "tokens": 120000, "cumulative_loss_tokens": 37800000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.964, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 302}, "mem_gb": 15.81} +{"step": 316, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22781831327453256, "tokens": 120000, "cumulative_loss_tokens": 37920000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 342.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.9, "frames": {"chat": 350}, "mem_gb": 15.72} +{"step": 317, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1352445625303158, "tokens": 120000, "cumulative_loss_tokens": 38040000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 411.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 292}, "mem_gb": 15.84} +{"step": 318, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2386181419883389, "tokens": 120000, "cumulative_loss_tokens": 38160000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 337.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.7, "frames": {"chat": 356}, "mem_gb": 15.65} +{"step": 319, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15787576438959222, "tokens": 120000, "cumulative_loss_tokens": 38280000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 345.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 347}, "mem_gb": 15.82} +{"step": 320, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1460279316786987, "tokens": 120000, "cumulative_loss_tokens": 38400000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.976, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 330}, "mem_gb": 16.01} +[eval step 320] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix \\(A\\) as:\n\\[ A = \\begin{pmatrix}\n12 & -1" +{"step": 321, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23903207845209787, "tokens": 120000, "cumulative_loss_tokens": 38520000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.6, "frames": {"chat": 363}, "mem_gb": 15.87} +{"step": 322, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21984903761375074, "tokens": 120000, "cumulative_loss_tokens": 38640000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.1, "frames": {"chat": 354}, "mem_gb": 15.73} +{"step": 323, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11327603132415873, "tokens": 120000, "cumulative_loss_tokens": 38760000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.93, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 270}, "mem_gb": 15.99} +{"step": 324, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22907125133417236, "tokens": 120000, "cumulative_loss_tokens": 38880000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 349.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.2, "frames": {"chat": 343}, "mem_gb": 15.77} +{"step": 325, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.051663052826110896, "tokens": 120000, "cumulative_loss_tokens": 39000000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.5, "frames": {"chat": 232}, "mem_gb": 15.96} +{"step": 326, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2181959554275653, "tokens": 120000, "cumulative_loss_tokens": 39120000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 346.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 346}, "mem_gb": 15.91} +{"step": 327, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16903356919861398, "tokens": 120000, "cumulative_loss_tokens": 39240000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.951, "comp_len": 392.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 306}, "mem_gb": 15.99} +{"step": 328, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10973309833916525, "tokens": 120000, "cumulative_loss_tokens": 39360000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 439.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 273}, "mem_gb": 16.04} +{"step": 329, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11865625682111519, "tokens": 120000, "cumulative_loss_tokens": 39480000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 451.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 266}, "mem_gb": 15.86} +{"step": 330, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1848333759183685, "tokens": 120000, "cumulative_loss_tokens": 39600000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.936, "comp_len": 401.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 299}, "mem_gb": 16.06} +[eval step 330] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given 4x4 matrix \\( A \\) as follows:\n\\[ A = \\begin{bm" +{"step": 331, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.123552789839264, "tokens": 120000, "cumulative_loss_tokens": 39720000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.88, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 259}, "mem_gb": 16.14} +{"step": 332, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14954215596408274, "tokens": 120000, "cumulative_loss_tokens": 39840000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.966, "comp_len": 369.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 325}, "mem_gb": 15.88} +{"step": 333, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1490584114160699, "tokens": 120000, "cumulative_loss_tokens": 39960000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 271}, "mem_gb": 15.95} +{"step": 334, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22186615592606054, "tokens": 120000, "cumulative_loss_tokens": 40080000, "grad_norm": 0.474609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 334.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.7, "frames": {"chat": 359}, "mem_gb": 15.74} +{"step": 335, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2153005481238865, "tokens": 120000, "cumulative_loss_tokens": 40200000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 347.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.6, "frames": {"chat": 345}, "mem_gb": 15.8} +{"step": 336, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2417936733730603, "tokens": 120000, "cumulative_loss_tokens": 40320000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 301.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.7, "frames": {"chat": 398}, "mem_gb": 15.83} +{"step": 337, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1374059521160554, "tokens": 120000, "cumulative_loss_tokens": 40440000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.964, "comp_len": 390.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 307}, "mem_gb": 16.03} +{"step": 338, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2298686686015067, "tokens": 120000, "cumulative_loss_tokens": 40560000, "grad_norm": 0.48828125, "lr": 3e-05, "finish_rate": 0.995, "comp_len": 308.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.8, "frames": {"chat": 389}, "mem_gb": 15.8} +{"step": 339, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03598521605608985, "tokens": 120000, "cumulative_loss_tokens": 40680000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.1, "frames": {"chat": 231}, "mem_gb": 16.04} +{"step": 340, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21775738089157579, "tokens": 120000, "cumulative_loss_tokens": 40800000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 373.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 321}, "mem_gb": 15.71} +[eval step 340] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly independent rows or col' +{"step": 341, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.25866739971299346, "tokens": 120000, "cumulative_loss_tokens": 40920000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 357.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.2, "frames": {"chat": 336}, "mem_gb": 15.73} +{"step": 342, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23178869430430543, "tokens": 120000, "cumulative_loss_tokens": 41040000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 355.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.0, "frames": {"chat": 338}, "mem_gb": 15.89} +{"step": 343, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12498800429821325, "tokens": 120000, "cumulative_loss_tokens": 41160000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.945, "comp_len": 436.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 275}, "mem_gb": 16.03} +{"step": 344, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08653919376684353, "tokens": 120000, "cumulative_loss_tokens": 41280000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.0, "frames": {"chat": 226}, "mem_gb": 16.05} +{"step": 345, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24881534034009092, "tokens": 120000, "cumulative_loss_tokens": 41400000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 346.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.8, "frames": {"chat": 346}, "mem_gb": 15.85} +{"step": 346, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23086289331723625, "tokens": 120000, "cumulative_loss_tokens": 41520000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.7, "frames": {"chat": 363}, "mem_gb": 15.66} +{"step": 347, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1248968276354329, "tokens": 120000, "cumulative_loss_tokens": 41640000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.955, "comp_len": 383.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 313}, "mem_gb": 15.76} +{"step": 348, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.26007929761900256, "tokens": 120000, "cumulative_loss_tokens": 41760000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 323.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.4, "frames": {"chat": 371}, "mem_gb": 15.82} +{"step": 349, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18477537309393907, "tokens": 120000, "cumulative_loss_tokens": 41880000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 373.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 321}, "mem_gb": 15.6} +{"step": 350, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2277006753143544, "tokens": 120000, "cumulative_loss_tokens": 42000000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 365.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.9, "frames": {"chat": 328}, "mem_gb": 15.68} +[eval step 350] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n12 &" +{"step": 351, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30436301054852083, "tokens": 120000, "cumulative_loss_tokens": 42120000, "grad_norm": 0.53125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 357.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 336}, "mem_gb": 15.9} +{"step": 352, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23449951709414987, "tokens": 120000, "cumulative_loss_tokens": 42240000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 323.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.9, "frames": {"chat": 371}, "mem_gb": 15.79} +{"step": 353, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0563998339336055, "tokens": 120000, "cumulative_loss_tokens": 42360000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.4, "frames": {"chat": 226}, "mem_gb": 16.03} +{"step": 354, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21147021062457935, "tokens": 120000, "cumulative_loss_tokens": 42480000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.931, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 304}, "mem_gb": 16.04} +{"step": 355, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05292390454676934, "tokens": 120000, "cumulative_loss_tokens": 42600000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.7, "frames": {"chat": 217}, "mem_gb": 16.03} +{"step": 356, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20416466785751594, "tokens": 120000, "cumulative_loss_tokens": 42720000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 351.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 341}, "mem_gb": 15.72} +{"step": 357, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2267281416730645, "tokens": 120000, "cumulative_loss_tokens": 42840000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 343.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.9, "frames": {"chat": 349}, "mem_gb": 15.62} +{"step": 358, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.25760618139517805, "tokens": 120000, "cumulative_loss_tokens": 42960000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 336.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.9, "frames": {"chat": 357}, "mem_gb": 15.82} +{"step": 359, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2787673988107902, "tokens": 120000, "cumulative_loss_tokens": 43080000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 325.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.0, "frames": {"chat": 369}, "mem_gb": 15.6} +{"step": 360, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14880364139573649, "tokens": 120000, "cumulative_loss_tokens": 43200000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.974, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 302}, "mem_gb": 15.88} +[eval step 360] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n1" +{"step": 361, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14683144709807822, "tokens": 120000, "cumulative_loss_tokens": 43320000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.968, "comp_len": 385.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 311}, "mem_gb": 15.7} +{"step": 362, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.25131194620172803, "tokens": 120000, "cumulative_loss_tokens": 43440000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 365.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 328}, "mem_gb": 15.78} +{"step": 363, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2766481215565776, "tokens": 120000, "cumulative_loss_tokens": 43560000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.6, "frames": {"chat": 353}, "mem_gb": 15.82} +{"step": 364, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20890475435064484, "tokens": 120000, "cumulative_loss_tokens": 43680000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 360.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 333}, "mem_gb": 15.89} +{"step": 365, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15655357416563978, "tokens": 120000, "cumulative_loss_tokens": 43800000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.966, "comp_len": 375.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 320}, "mem_gb": 15.62} +{"step": 366, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07142020008829422, "tokens": 120000, "cumulative_loss_tokens": 43920000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.7, "frames": {"chat": 217}, "mem_gb": 16.13} +{"step": 367, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.042445074440181876, "tokens": 120000, "cumulative_loss_tokens": 44040000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.849, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.1, "frames": {"chat": 219}, "mem_gb": 16.05} +{"step": 368, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15583300799441835, "tokens": 120000, "cumulative_loss_tokens": 44160000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.905, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 262}, "mem_gb": 15.96} +{"step": 369, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2563023217106859, "tokens": 120000, "cumulative_loss_tokens": 44280000, "grad_norm": 0.53125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 320.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.6, "frames": {"chat": 375}, "mem_gb": 15.85} +{"step": 370, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21440070870715813, "tokens": 120000, "cumulative_loss_tokens": 44400000, "grad_norm": 0.474609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.0, "frames": {"chat": 342}, "mem_gb": 15.7} +[eval step 370] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given 4x4 matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4" +{"step": 371, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12221701149945147, "tokens": 120000, "cumulative_loss_tokens": 44520000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.945, "comp_len": 389.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 308}, "mem_gb": 16.01} +{"step": 372, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.030700478985750426, "tokens": 120000, "cumulative_loss_tokens": 44640000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.1, "frames": {"chat": 211}, "mem_gb": 16.04} +{"step": 373, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0886014010820693, "tokens": 120000, "cumulative_loss_tokens": 44760000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 427.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 281}, "mem_gb": 16.04} +{"step": 374, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19356555437323017, "tokens": 120000, "cumulative_loss_tokens": 44880000, "grad_norm": 0.546875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 327.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.4, "frames": {"chat": 367}, "mem_gb": 15.59} +{"step": 375, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19524073943241188, "tokens": 120000, "cumulative_loss_tokens": 45000000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 384.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 312}, "mem_gb": 15.75} +{"step": 376, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1506790350826923, "tokens": 120000, "cumulative_loss_tokens": 45120000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 271}, "mem_gb": 16.06} +{"step": 377, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09377165296597717, "tokens": 120000, "cumulative_loss_tokens": 45240000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 446.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 269}, "mem_gb": 16.03} +{"step": 378, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1866174870405967, "tokens": 120000, "cumulative_loss_tokens": 45360000, "grad_norm": 0.54296875, "lr": 3e-05, "finish_rate": 0.989, "comp_len": 340.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 352}, "mem_gb": 15.55} +{"step": 379, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0727388446987141, "tokens": 120000, "cumulative_loss_tokens": 45480000, "grad_norm": 0.298828125, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 243}, "mem_gb": 16.04} +{"step": 380, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12049580976279298, "tokens": 120000, "cumulative_loss_tokens": 45600000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.952, "comp_len": 409.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 293}, "mem_gb": 15.96} +[eval step 380] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly indepe' +{"step": 381, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06266305668996647, "tokens": 120000, "cumulative_loss_tokens": 45720000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.872, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 242}, "mem_gb": 16.04} +{"step": 382, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20161160860074062, "tokens": 120000, "cumulative_loss_tokens": 45840000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.9, "frames": {"chat": 353}, "mem_gb": 15.73} +{"step": 383, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19306930440085318, "tokens": 120000, "cumulative_loss_tokens": 45960000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 355.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.0, "frames": {"chat": 338}, "mem_gb": 15.74} +{"step": 384, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09298814616055849, "tokens": 120000, "cumulative_loss_tokens": 46080000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 259}, "mem_gb": 16.04} +{"step": 385, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18186330963566433, "tokens": 120000, "cumulative_loss_tokens": 46200000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 311.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.0, "frames": {"chat": 385}, "mem_gb": 15.59} +{"step": 386, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16286379592275868, "tokens": 120000, "cumulative_loss_tokens": 46320000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 338.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.8, "frames": {"chat": 355}, "mem_gb": 15.76} +{"step": 387, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1550285114251077, "tokens": 120000, "cumulative_loss_tokens": 46440000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 365.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 328}, "mem_gb": 15.66} +{"step": 388, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16467833463284187, "tokens": 120000, "cumulative_loss_tokens": 46560000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 338.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.8, "frames": {"chat": 355}, "mem_gb": 15.81} +{"step": 389, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20514879045716175, "tokens": 120000, "cumulative_loss_tokens": 46680000, "grad_norm": 0.490234375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 327.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.4, "frames": {"chat": 367}, "mem_gb": 15.68} +{"step": 390, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17667803509252455, "tokens": 120000, "cumulative_loss_tokens": 46800000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 349.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.6, "frames": {"chat": 343}, "mem_gb": 15.72} +[eval step 390] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given 4x4 matrix as \\( A \\):\n\n\\[ A = \\begin{pmatrix}\n" +{"step": 391, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03950484952997261, "tokens": 120000, "cumulative_loss_tokens": 46920000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.7, "frames": {"chat": 244}, "mem_gb": 16.04} +{"step": 392, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11025849760763813, "tokens": 120000, "cumulative_loss_tokens": 47040000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 354.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 339}, "mem_gb": 15.88} +{"step": 393, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10967219756064005, "tokens": 120000, "cumulative_loss_tokens": 47160000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 402.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 298}, "mem_gb": 16.03} +{"step": 394, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1773800556027641, "tokens": 120000, "cumulative_loss_tokens": 47280000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 301.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.6, "frames": {"chat": 398}, "mem_gb": 15.91} +{"step": 395, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16633155678164524, "tokens": 120000, "cumulative_loss_tokens": 47400000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 331.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.9, "frames": {"chat": 362}, "mem_gb": 16.01} +{"step": 396, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11098843370222021, "tokens": 120000, "cumulative_loss_tokens": 47520000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 383.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 313}, "mem_gb": 15.88} +{"step": 397, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20024378271650833, "tokens": 120000, "cumulative_loss_tokens": 47640000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 370.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 324}, "mem_gb": 15.79} +{"step": 398, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11992832584790886, "tokens": 120000, "cumulative_loss_tokens": 47760000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.936, "comp_len": 382.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 314}, "mem_gb": 15.85} +{"step": 399, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2123284807103686, "tokens": 120000, "cumulative_loss_tokens": 47880000, "grad_norm": 0.65625, "lr": 3e-05, "finish_rate": 0.991, "comp_len": 347.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 345}, "mem_gb": 15.77} +{"step": 400, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08385536621418627, "tokens": 120000, "cumulative_loss_tokens": 48000000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.897, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 262}, "mem_gb": 15.95} +[eval step 400] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly indepe' +{"step": 401, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21703253002436831, "tokens": 120000, "cumulative_loss_tokens": 48120000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 321.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.1, "frames": {"chat": 373}, "mem_gb": 15.75} +{"step": 402, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17621628788031327, "tokens": 120000, "cumulative_loss_tokens": 48240000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 359.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 334}, "mem_gb": 15.74} +{"step": 403, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12801904136755815, "tokens": 120000, "cumulative_loss_tokens": 48360000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 367.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 327}, "mem_gb": 15.89} +{"step": 404, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.027625936167376738, "tokens": 120000, "cumulative_loss_tokens": 48480000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.4, "frames": {"chat": 238}, "mem_gb": 16.06} +{"step": 405, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15306450330043833, "tokens": 120000, "cumulative_loss_tokens": 48600000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.961, "comp_len": 387.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.9, "frames": {"chat": 310}, "mem_gb": 15.79} +{"step": 406, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18156374052620183, "tokens": 120000, "cumulative_loss_tokens": 48720000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 345.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.3, "frames": {"chat": 347}, "mem_gb": 15.79} +{"step": 407, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17691095505828658, "tokens": 120000, "cumulative_loss_tokens": 48840000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 304}, "mem_gb": 15.72} +{"step": 408, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16285859790768784, "tokens": 120000, "cumulative_loss_tokens": 48960000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 349.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.7, "frames": {"chat": 343}, "mem_gb": 15.74} +{"step": 409, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20077014775959154, "tokens": 120000, "cumulative_loss_tokens": 49080000, "grad_norm": 0.5546875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 382.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 314}, "mem_gb": 15.73} +{"step": 410, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17390328064192243, "tokens": 120000, "cumulative_loss_tokens": 49200000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.97, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 300}, "mem_gb": 15.84} +[eval step 410] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n1" +{"step": 411, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13172015643895915, "tokens": 120000, "cumulative_loss_tokens": 49320000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.981, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 318}, "mem_gb": 15.88} +{"step": 412, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09676262668388275, "tokens": 120000, "cumulative_loss_tokens": 49440000, "grad_norm": 0.3046875, "lr": 3e-05, "finish_rate": 0.939, "comp_len": 430.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 279}, "mem_gb": 15.91} +{"step": 413, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18304422753804053, "tokens": 120000, "cumulative_loss_tokens": 49560000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.984, "comp_len": 329.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.5, "frames": {"chat": 364}, "mem_gb": 15.98} +{"step": 414, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12250509955501183, "tokens": 120000, "cumulative_loss_tokens": 49680000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 256}, "mem_gb": 16.04} +{"step": 415, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.22661521423538217, "tokens": 120000, "cumulative_loss_tokens": 49800000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 352.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 340}, "mem_gb": 15.75} +{"step": 416, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09992620240498024, "tokens": 120000, "cumulative_loss_tokens": 49920000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.901, "comp_len": 424.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 283}, "mem_gb": 15.97} +{"step": 417, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18437018060648502, "tokens": 120000, "cumulative_loss_tokens": 50040000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 343.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.3, "frames": {"chat": 349}, "mem_gb": 15.76} +{"step": 418, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09829366903956203, "tokens": 120000, "cumulative_loss_tokens": 50160000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.963, "comp_len": 404.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 297}, "mem_gb": 15.83} +{"step": 419, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21936706093556324, "tokens": 120000, "cumulative_loss_tokens": 50280000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 334.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.6, "frames": {"chat": 359}, "mem_gb": 15.81} +{"step": 420, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13868955738450556, "tokens": 120000, "cumulative_loss_tokens": 50400000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.2, "frames": {"chat": 342}, "mem_gb": 15.85} +[eval step 420] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{pmatrix}\n12 &" +{"step": 421, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09069466766981253, "tokens": 120000, "cumulative_loss_tokens": 50520000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 446.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 269}, "mem_gb": 16.01} +{"step": 422, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13309008194206592, "tokens": 120000, "cumulative_loss_tokens": 50640000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.963, "comp_len": 370.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 324}, "mem_gb": 16.0} +{"step": 423, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10460173758862075, "tokens": 120000, "cumulative_loss_tokens": 50760000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.94, "comp_len": 427.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 281}, "mem_gb": 15.99} +{"step": 424, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1326145618700267, "tokens": 120000, "cumulative_loss_tokens": 50880000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.916, "comp_len": 418.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 287}, "mem_gb": 16.04} +{"step": 425, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1542503337391497, "tokens": 120000, "cumulative_loss_tokens": 51000000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.939, "comp_len": 405.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 296}, "mem_gb": 16.05} +{"step": 426, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10765284155028251, "tokens": 120000, "cumulative_loss_tokens": 51120000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 449.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 267}, "mem_gb": 16.1} +{"step": 427, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10783903675417726, "tokens": 120000, "cumulative_loss_tokens": 51240000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 249}, "mem_gb": 16.0} +{"step": 428, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2076204192122367, "tokens": 120000, "cumulative_loss_tokens": 51360000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.5, "frames": {"chat": 363}, "mem_gb": 15.7} +{"step": 429, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1255934642267103, "tokens": 120000, "cumulative_loss_tokens": 51480000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.953, "comp_len": 401.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 299}, "mem_gb": 15.84} +{"step": 430, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19177433499107138, "tokens": 120000, "cumulative_loss_tokens": 51600000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.5, "frames": {"chat": 331}, "mem_gb": 15.71} +[eval step 430] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given 4x4 matrix \\( A \\) as follows:\n\\[ A = \\begin" +{"step": 431, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09366897407051486, "tokens": 120000, "cumulative_loss_tokens": 51720000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.954, "comp_len": 428.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 280}, "mem_gb": 15.93} +{"step": 432, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2013886440964571, "tokens": 120000, "cumulative_loss_tokens": 51840000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.994, "comp_len": 342.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 350}, "mem_gb": 15.91} +{"step": 433, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1908259281904126, "tokens": 120000, "cumulative_loss_tokens": 51960000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 373.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 321}, "mem_gb": 15.67} +{"step": 434, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09085603848312361, "tokens": 120000, "cumulative_loss_tokens": 52080000, "grad_norm": 0.298828125, "lr": 3e-05, "finish_rate": 0.943, "comp_len": 427.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 281}, "mem_gb": 15.98} +{"step": 435, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21918390367375687, "tokens": 120000, "cumulative_loss_tokens": 52200000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.8, "frames": {"chat": 331}, "mem_gb": 15.65} +{"step": 436, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1565575725618129, "tokens": 120000, "cumulative_loss_tokens": 52320000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.956, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 318}, "mem_gb": 16.02} +{"step": 437, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11646176510953034, "tokens": 120000, "cumulative_loss_tokens": 52440000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.942, "comp_len": 388.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 309}, "mem_gb": 16.04} +{"step": 438, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12193210456989861, "tokens": 120000, "cumulative_loss_tokens": 52560000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.941, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 304}, "mem_gb": 15.93} +{"step": 439, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07701745812663188, "tokens": 120000, "cumulative_loss_tokens": 52680000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 430.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 279}, "mem_gb": 15.95} +{"step": 440, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.034134947031123256, "tokens": 120000, "cumulative_loss_tokens": 52800000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.7, "frames": {"chat": 248}, "mem_gb": 15.95} +[eval step 440] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n12 &" +{"step": 441, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15044653408117592, "tokens": 120000, "cumulative_loss_tokens": 52920000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.907, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 268}, "mem_gb": 15.92} +{"step": 442, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14706381488045833, "tokens": 120000, "cumulative_loss_tokens": 53040000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.953, "comp_len": 401.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 299}, "mem_gb": 16.02} +{"step": 443, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1639198791468361, "tokens": 120000, "cumulative_loss_tokens": 53160000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.978, "comp_len": 372.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.1, "frames": {"chat": 322}, "mem_gb": 15.89} +{"step": 444, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08993425388707159, "tokens": 120000, "cumulative_loss_tokens": 53280000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 248}, "mem_gb": 16.06} +{"step": 445, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14992557585640656, "tokens": 120000, "cumulative_loss_tokens": 53400000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 368.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 326}, "mem_gb": 15.95} +{"step": 446, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13199191171765948, "tokens": 120000, "cumulative_loss_tokens": 53520000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.984, "comp_len": 382.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 314}, "mem_gb": 15.71} +{"step": 447, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03281254557561285, "tokens": 120000, "cumulative_loss_tokens": 53640000, "grad_norm": 0.2451171875, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 232}, "mem_gb": 16.05} +{"step": 448, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20584237420060672, "tokens": 120000, "cumulative_loss_tokens": 53760000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.994, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.2, "frames": {"chat": 330}, "mem_gb": 15.78} +{"step": 449, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0928178978715092, "tokens": 120000, "cumulative_loss_tokens": 53880000, "grad_norm": 0.298828125, "lr": 3e-05, "finish_rate": 0.901, "comp_len": 411.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 292}, "mem_gb": 15.93} +{"step": 450, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15818013695871147, "tokens": 120000, "cumulative_loss_tokens": 54000000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.978, "comp_len": 384.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 312}, "mem_gb": 15.95} +[eval step 450] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly indepe' +checkpoint snapshot queued -> outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0450 +{"step": 451, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21172066032474396, "tokens": 120000, "cumulative_loss_tokens": 54120000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 314.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.2, "frames": {"chat": 382}, "mem_gb": 15.76} +{"step": 452, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.22152453744931458, "tokens": 120000, "cumulative_loss_tokens": 54240000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.994, "comp_len": 333.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.0, "frames": {"chat": 360}, "mem_gb": 15.63} +{"step": 453, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02605439814896478, "tokens": 120000, "cumulative_loss_tokens": 54360000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.2, "frames": {"chat": 222}, "mem_gb": 15.9} +{"step": 454, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20108578163626759, "tokens": 120000, "cumulative_loss_tokens": 54480000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.5, "frames": {"chat": 342}, "mem_gb": 15.94} +{"step": 455, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1972158631942235, "tokens": 120000, "cumulative_loss_tokens": 54600000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 356.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 337}, "mem_gb": 15.74} +{"step": 456, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11860149113883575, "tokens": 120000, "cumulative_loss_tokens": 54720000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.934, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 304}, "mem_gb": 16.0} +{"step": 457, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12831736822670792, "tokens": 120000, "cumulative_loss_tokens": 54840000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.954, "comp_len": 368.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.3, "frames": {"chat": 326}, "mem_gb": 15.89} +{"step": 458, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16882150728447984, "tokens": 120000, "cumulative_loss_tokens": 54960000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 387.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 310}, "mem_gb": 15.95} +{"step": 459, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1581653488818789, "tokens": 120000, "cumulative_loss_tokens": 55080000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.982, "comp_len": 365.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 328}, "mem_gb": 15.77} +{"step": 460, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19649457880739743, "tokens": 120000, "cumulative_loss_tokens": 55200000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 345.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.2, "frames": {"chat": 347}, "mem_gb": 15.6} +[eval step 460] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -' +{"step": 461, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1654393976194008, "tokens": 120000, "cumulative_loss_tokens": 55320000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.976, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 330}, "mem_gb": 15.89} +{"step": 462, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0355358585749132, "tokens": 120000, "cumulative_loss_tokens": 55440000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.751, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.9, "frames": {"chat": 209}, "mem_gb": 16.09} +{"step": 463, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19095296816974877, "tokens": 120000, "cumulative_loss_tokens": 55560000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.99, "comp_len": 300.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.9, "frames": {"chat": 399}, "mem_gb": 15.67} +{"step": 464, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15986459132861347, "tokens": 120000, "cumulative_loss_tokens": 55680000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 341.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.1, "frames": {"chat": 351}, "mem_gb": 15.75} +{"step": 465, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11593475418574332, "tokens": 120000, "cumulative_loss_tokens": 55800000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.975, "comp_len": 372.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 322}, "mem_gb": 15.75} +{"step": 466, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18452940405049206, "tokens": 120000, "cumulative_loss_tokens": 55920000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 360.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 333}, "mem_gb": 15.64} +{"step": 467, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13806939297091836, "tokens": 120000, "cumulative_loss_tokens": 56040000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.955, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 330}, "mem_gb": 15.93} +{"step": 468, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.026204069481013965, "tokens": 120000, "cumulative_loss_tokens": 56160000, "grad_norm": 0.1953125, "lr": 3e-05, "finish_rate": 0.866, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.2, "frames": {"chat": 238}, "mem_gb": 15.96} +{"step": 469, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08274232047294887, "tokens": 120000, "cumulative_loss_tokens": 56280000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.943, "comp_len": 404.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 297}, "mem_gb": 15.87} +{"step": 470, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17612028898630913, "tokens": 120000, "cumulative_loss_tokens": 56400000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.991, "comp_len": 347.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 345}, "mem_gb": 15.83} +[eval step 470] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n1" +{"step": 471, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12068462449768558, "tokens": 120000, "cumulative_loss_tokens": 56520000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.944, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 304}, "mem_gb": 15.98} +{"step": 472, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16559650549436145, "tokens": 120000, "cumulative_loss_tokens": 56640000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.982, "comp_len": 368.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 326}, "mem_gb": 15.98} +{"step": 473, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1939527134188451, "tokens": 120000, "cumulative_loss_tokens": 56760000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 345.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 347}, "mem_gb": 15.91} +{"step": 474, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18024646681703937, "tokens": 120000, "cumulative_loss_tokens": 56880000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 369.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 325}, "mem_gb": 15.79} +{"step": 475, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1052100208830166, "tokens": 120000, "cumulative_loss_tokens": 57000000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.933, "comp_len": 401.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 299}, "mem_gb": 15.98} +{"step": 476, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2246520654431486, "tokens": 120000, "cumulative_loss_tokens": 57120000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 347.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.8, "frames": {"chat": 345}, "mem_gb": 15.76} +{"step": 477, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1814513569199325, "tokens": 120000, "cumulative_loss_tokens": 57240000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 348.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 344}, "mem_gb": 15.65} +{"step": 478, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13653103163487587, "tokens": 120000, "cumulative_loss_tokens": 57360000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.985, "comp_len": 348.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 344}, "mem_gb": 15.91} +{"step": 479, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1226733575352002, "tokens": 120000, "cumulative_loss_tokens": 57480000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.962, "comp_len": 416.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 288}, "mem_gb": 15.98} +{"step": 480, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11710733379688269, "tokens": 120000, "cumulative_loss_tokens": 57600000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 396.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 303}, "mem_gb": 15.68} +[eval step 480] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n1" +{"step": 481, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07719322284079777, "tokens": 120000, "cumulative_loss_tokens": 57720000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.898, "comp_len": 422.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 284}, "mem_gb": 15.87} +{"step": 482, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21449646458970384, "tokens": 120000, "cumulative_loss_tokens": 57840000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 325.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.4, "frames": {"chat": 369}, "mem_gb": 15.77} +{"step": 483, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09005961197530074, "tokens": 120000, "cumulative_loss_tokens": 57960000, "grad_norm": 0.3046875, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 446.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 269}, "mem_gb": 16.02} +{"step": 484, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19965751514827523, "tokens": 120000, "cumulative_loss_tokens": 58080000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 351.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.8, "frames": {"chat": 341}, "mem_gb": 15.78} +{"step": 485, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19275502525821794, "tokens": 120000, "cumulative_loss_tokens": 58200000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 312.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.1, "frames": {"chat": 384}, "mem_gb": 15.55} +{"step": 486, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17101380572033426, "tokens": 120000, "cumulative_loss_tokens": 58320000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.3, "frames": {"chat": 358}, "mem_gb": 15.71} +{"step": 487, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1823475890875173, "tokens": 120000, "cumulative_loss_tokens": 58440000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.9, "frames": {"chat": 342}, "mem_gb": 15.89} +{"step": 488, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20596611784929408, "tokens": 120000, "cumulative_loss_tokens": 58560000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 328.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.5, "frames": {"chat": 365}, "mem_gb": 15.71} +{"step": 489, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18416945955871294, "tokens": 120000, "cumulative_loss_tokens": 58680000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 356.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.4, "frames": {"chat": 337}, "mem_gb": 15.94} +{"step": 490, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16407542195774924, "tokens": 120000, "cumulative_loss_tokens": 58800000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.961, "comp_len": 360.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.7, "frames": {"chat": 333}, "mem_gb": 16.04} +[eval step 490] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n12 &" +{"step": 491, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11082636141755307, "tokens": 120000, "cumulative_loss_tokens": 58920000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 409.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 293}, "mem_gb": 16.08} +{"step": 492, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09701780941213947, "tokens": 120000, "cumulative_loss_tokens": 59040000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.973, "comp_len": 412.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 291}, "mem_gb": 15.84} +{"step": 493, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1673443554365076, "tokens": 120000, "cumulative_loss_tokens": 59160000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 351.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.5, "frames": {"chat": 341}, "mem_gb": 15.88} +{"step": 494, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10571547078011402, "tokens": 120000, "cumulative_loss_tokens": 59280000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.97, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 300}, "mem_gb": 15.81} +{"step": 495, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09014458696002452, "tokens": 120000, "cumulative_loss_tokens": 59400000, "grad_norm": 0.298828125, "lr": 3e-05, "finish_rate": 0.885, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 270}, "mem_gb": 16.03} +{"step": 496, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07189943271749653, "tokens": 120000, "cumulative_loss_tokens": 59520000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.7, "frames": {"chat": 259}, "mem_gb": 16.05} +{"step": 497, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.182238545052226, "tokens": 120000, "cumulative_loss_tokens": 59640000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 338.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 355}, "mem_gb": 15.63} +{"step": 498, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20417358451887654, "tokens": 120000, "cumulative_loss_tokens": 59760000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 309.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.4, "frames": {"chat": 388}, "mem_gb": 15.98} +{"step": 499, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16829194593451297, "tokens": 120000, "cumulative_loss_tokens": 59880000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 318}, "mem_gb": 15.7} +{"step": 500, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1956629434169891, "tokens": 120000, "cumulative_loss_tokens": 60000000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 313.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.0, "frames": {"chat": 383}, "mem_gb": 15.83} +[eval step 500] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given 4x4 matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4" +checkpoint snapshot queued -> outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0500 +wandb: updating run metadata +wandb: uploading config.yaml; uploading output.log; uploading wandb-summary.json +wandb: uploading summary +wandb: +wandb: Run history: +wandb: comp_len ▆▂▁█▆▄▃▄▇▄▃█▅▂▆▅▅▂▅▄▃▅▄▅▅▃▄▃▅▂▁▂▃▃▃▃▃▄▄▆ +wandb: cumulative_loss_tokens ▁▁▁▁▁▁▁▂▂▂▃▃▃▄▄▄▄▅▅▅▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇█████ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅███████████ +wandb: finish_rate ▆▄▇▅▆████▇▁██▆▇█▂█▆██▅██▇█▃▇▇▇▆█████▄██▇ +wandb: forward_topk_kl ▇█▇▃▇▆▄▆▄▆▆▆▄▄▅▆▄▆▆▄▁▃▂▄▄▃▂▃▃▃▄▃▂▄▂▃▁▃▃▃ +wandb: grad_norm █▂▂▁▂▁▂▂▂▁▂▁▁▂▂▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▂▁▁▁▁▁▁▁▁ +wandb: lr ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: mem_gb ▇▇▇▇▅▄▃▄▇▇▂█▄▆▃█▅▇▆▂▃▄▇▃▆▇▃▄▆▅▁▁▁▅▃▃▄▇▆▃ +wandb: step ▁▁▁▁▂▂▂▃▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇▇▇▇▇█ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 313.3 +wandb: cumulative_loss_tokens 60000000 +wandb: epoch 2 +wandb: finish_rate 0.997 +wandb: forward_topk_kl 0.19566 +wandb: grad_norm 0.41406 +wandb: lr 3e-05 +wandb: mem_gb 15.83 +wandb: step 500 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run glean_keep50_s1224_long500 at: https://wandb.ai/hbfreed/glean-general-grid/runs/77wxfx9w +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-general-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/wandb/run-20260719_142238-77wxfx9w/logs diff --git a/healed/grid_general_fairness/glean_keep50_s1224_long500.eval.log b/healed/grid_general_fairness/glean_keep50_s1224_long500.eval.log new file mode 100644 index 0000000000000000000000000000000000000000..80451c969e50f30aa1cc165acea94ca1686cd3e5 --- /dev/null +++ b/healed/grid_general_fairness/glean_keep50_s1224_long500.eval.log @@ -0,0 +1,141 @@ +2026-07-19T21:26:45-07:00 serving outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0150 on GPU 2 port 8422 +2026-07-19T21:26:45-07:00 waiting for server /health ... +2026-07-19T21:27:20-07:00 server up; chat pass [gsm8k_cot_zeroshot,minerva_math500,ifeval] +2026-07-19:21:27:27 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot', 'minerva_math500', 'ifeval'] +2026-07-19:21:27:29 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234 +2026-07-19:21:27:29 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding! +2026-07-19:21:27:29 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8422/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3} +2026-07-19:21:27:29 INFO [models.api_models:179] Using max length 2048 - 1 +2026-07-19:21:27:29 INFO [models.api_models:200] Using tokenizer None +2026-07-19:21:27:34 INFO [evaluator_utils:446] Selected tasks: +2026-07-19:21:27:34 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml) +2026-07-19:21:27:34 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml) +2026-07-19:21:27:34 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml) +2026-07-19:21:27:34 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280} +2026-07-19:21:27:34 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-19:21:27:34 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-19:21:27:34 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0... + 0%| | 0/1319 [00:00 outputs/evals/general_suite/healed/glean_keep50_s1224_long500_step150 +2026-07-19T21:40:19-07:00 serving outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0500 on GPU 2 port 8422 +2026-07-19T21:40:19-07:00 waiting for server /health ... +2026-07-19T21:40:49-07:00 server up; chat pass [gsm8k_cot_zeroshot,minerva_math500,ifeval] +2026-07-19:21:40:56 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot', 'minerva_math500', 'ifeval'] +2026-07-19:21:40:58 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234 +2026-07-19:21:40:58 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding! +2026-07-19:21:40:58 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8422/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3} +2026-07-19:21:40:58 INFO [models.api_models:179] Using max length 2048 - 1 +2026-07-19:21:40:58 INFO [models.api_models:200] Using tokenizer None +2026-07-19:21:41:03 INFO [evaluator_utils:446] Selected tasks: +2026-07-19:21:41:03 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml) +2026-07-19:21:41:03 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml) +2026-07-19:21:41:03 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml) +2026-07-19:21:41:03 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280} +2026-07-19:21:41:03 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-19:21:41:03 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-19:21:41:03 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0... + 0%| | 0/1319 [00:00 outputs/evals/general_suite/healed/glean_keep50_s1224_long500_step500 diff --git a/healed/grid_general_fairness/reap_keep50_s1224_long500.console.log b/healed/grid_general_fairness/reap_keep50_s1224_long500.console.log new file mode 100644 index 0000000000000000000000000000000000000000..cd3b2991c0a1c5140128601aaa9da3b766908d9d --- /dev/null +++ b/healed/grid_general_fairness/reap_keep50_s1224_long500.console.log @@ -0,0 +1,602 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run vehg5sig +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_general_fairness/reap_keep50_s1224_long500/wandb/run-20260719_142008-vehg5sig +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run reap_keep50_s1224_long500 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-general-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-general-grid/runs/vehg5sig + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/grid_general_fairness/reap_keep50_s1224_long500/step0150 +{"step": 151, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4238436554264277, "tokens": 120000, "cumulative_loss_tokens": 18120000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 317.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.2, "frames": {"chat": 378}, "mem_gb": 15.78} +{"step": 152, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5016522423449283, "tokens": 120000, "cumulative_loss_tokens": 18240000, "grad_norm": 0.62890625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 326.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.3, "frames": {"chat": 368}, "mem_gb": 15.88} +{"step": 153, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21816937528620475, "tokens": 120000, "cumulative_loss_tokens": 18360000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.898, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 256}, "mem_gb": 15.99} +{"step": 154, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24708353220367182, "tokens": 120000, "cumulative_loss_tokens": 18480000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.93, "comp_len": 381.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.8, "frames": {"chat": 315}, "mem_gb": 15.93} +{"step": 155, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2387591682672811, "tokens": 120000, "cumulative_loss_tokens": 18600000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.974, "comp_len": 384.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.2, "frames": {"chat": 312}, "mem_gb": 15.86} +{"step": 156, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3530088965745022, "tokens": 120000, "cumulative_loss_tokens": 18720000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.97, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.8, "frames": {"chat": 330}, "mem_gb": 16.04} +{"step": 157, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.30485207046639795, "tokens": 120000, "cumulative_loss_tokens": 18840000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.954, "comp_len": 367.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.9, "frames": {"chat": 327}, "mem_gb": 16.05} +{"step": 158, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4054355481557548, "tokens": 120000, "cumulative_loss_tokens": 18960000, "grad_norm": 0.60546875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 341.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.5, "frames": {"chat": 351}, "mem_gb": 15.7} +{"step": 159, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23768405785240854, "tokens": 120000, "cumulative_loss_tokens": 19080000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.973, "comp_len": 357.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.5, "frames": {"chat": 336}, "mem_gb": 15.75} +{"step": 160, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.41708124500568955, "tokens": 120000, "cumulative_loss_tokens": 19200000, "grad_norm": 0.58984375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 332.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 70.0, "frames": {"chat": 361}, "mem_gb": 15.73} +[eval step 160] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given 4x4 matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & " +{"step": 161, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.42883908391973624, "tokens": 120000, "cumulative_loss_tokens": 19320000, "grad_norm": 0.5859375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 328.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 71.4, "frames": {"chat": 365}, "mem_gb": 15.67} +{"step": 162, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.39819961881730703, "tokens": 120000, "cumulative_loss_tokens": 19440000, "grad_norm": 0.5546875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 364.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.3, "frames": {"chat": 329}, "mem_gb": 15.73} +{"step": 163, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.26645059775626284, "tokens": 120000, "cumulative_loss_tokens": 19560000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.931, "comp_len": 416.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.6, "frames": {"chat": 288}, "mem_gb": 15.94} +{"step": 164, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20206951542422175, "tokens": 120000, "cumulative_loss_tokens": 19680000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 451.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.5, "frames": {"chat": 266}, "mem_gb": 16.09} +{"step": 165, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3668308581629147, "tokens": 120000, "cumulative_loss_tokens": 19800000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 372.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.5, "frames": {"chat": 322}, "mem_gb": 16.0} +{"step": 166, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.35375437433080126, "tokens": 120000, "cumulative_loss_tokens": 19920000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.989, "comp_len": 334.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.8, "frames": {"chat": 359}, "mem_gb": 15.7} +{"step": 167, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3022041524942343, "tokens": 120000, "cumulative_loss_tokens": 20040000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 385.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.1, "frames": {"chat": 311}, "mem_gb": 15.83} +{"step": 168, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2680534577645982, "tokens": 120000, "cumulative_loss_tokens": 20160000, "grad_norm": 0.48828125, "lr": 3e-05, "finish_rate": 0.95, "comp_len": 396.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.3, "frames": {"chat": 303}, "mem_gb": 15.88} +{"step": 169, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4362435298827787, "tokens": 120000, "cumulative_loss_tokens": 20280000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 351.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.8, "frames": {"chat": 341}, "mem_gb": 15.89} +{"step": 170, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.11099931483801144, "tokens": 120000, "cumulative_loss_tokens": 20400000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 229}, "mem_gb": 15.95} +[eval step 170] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1' +{"step": 171, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2790184115679003, "tokens": 120000, "cumulative_loss_tokens": 20520000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.964, "comp_len": 389.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.2, "frames": {"chat": 308}, "mem_gb": 16.11} +{"step": 172, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.38425825963026533, "tokens": 120000, "cumulative_loss_tokens": 20640000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 331.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.2, "frames": {"chat": 362}, "mem_gb": 15.61} +{"step": 173, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.40624842363204805, "tokens": 120000, "cumulative_loss_tokens": 20760000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 315.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 75.4, "frames": {"chat": 380}, "mem_gb": 15.85} +{"step": 174, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.38374614504699905, "tokens": 120000, "cumulative_loss_tokens": 20880000, "grad_norm": 0.5859375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 336.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.3, "frames": {"chat": 357}, "mem_gb": 15.63} +{"step": 175, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2148200512635211, "tokens": 120000, "cumulative_loss_tokens": 21000000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 472.4, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 55.1, "frames": {"chat": 254}, "mem_gb": 16.11} +{"step": 176, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17002530116820708, "tokens": 120000, "cumulative_loss_tokens": 21120000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.874, "comp_len": 459.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.1, "frames": {"chat": 261}, "mem_gb": 16.07} +{"step": 177, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.36013852669422824, "tokens": 120000, "cumulative_loss_tokens": 21240000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.982, "comp_len": 431.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.8, "frames": {"chat": 278}, "mem_gb": 15.87} +{"step": 178, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3871273392839823, "tokens": 120000, "cumulative_loss_tokens": 21360000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 360.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.7, "frames": {"chat": 333}, "mem_gb": 16.0} +{"step": 179, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19604582638765375, "tokens": 120000, "cumulative_loss_tokens": 21480000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.8, "frames": {"chat": 250}, "mem_gb": 15.97} +{"step": 180, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.46790155845905346, "tokens": 120000, "cumulative_loss_tokens": 21600000, "grad_norm": 0.63671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 336.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.1, "frames": {"chat": 357}, "mem_gb": 15.74} +[eval step 180] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1' +{"step": 181, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.46934490084790936, "tokens": 120000, "cumulative_loss_tokens": 21720000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.9, "frames": {"chat": 330}, "mem_gb": 15.71} +{"step": 182, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.41195430842408287, "tokens": 120000, "cumulative_loss_tokens": 21840000, "grad_norm": 0.58984375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 341.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.1, "frames": {"chat": 351}, "mem_gb": 15.75} +{"step": 183, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27260757924347806, "tokens": 120000, "cumulative_loss_tokens": 21960000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 413.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.9, "frames": {"chat": 290}, "mem_gb": 15.88} +{"step": 184, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.34496029592836275, "tokens": 120000, "cumulative_loss_tokens": 22080000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 372.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.6, "frames": {"chat": 322}, "mem_gb": 15.82} +{"step": 185, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1899029776233559, "tokens": 120000, "cumulative_loss_tokens": 22200000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 451.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.8, "frames": {"chat": 266}, "mem_gb": 16.06} +{"step": 186, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3483811936741074, "tokens": 120000, "cumulative_loss_tokens": 22320000, "grad_norm": 0.59375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.8, "frames": {"chat": 331}, "mem_gb": 15.91} +{"step": 187, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3247120169549249, "tokens": 120000, "cumulative_loss_tokens": 22440000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.3, "frames": {"chat": 342}, "mem_gb": 15.67} +{"step": 188, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3731004893165392, "tokens": 120000, "cumulative_loss_tokens": 22560000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.0, "frames": {"chat": 358}, "mem_gb": 15.71} +{"step": 189, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3187536304532861, "tokens": 120000, "cumulative_loss_tokens": 22680000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 348.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.8, "frames": {"chat": 344}, "mem_gb": 15.79} +{"step": 190, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20411863568654906, "tokens": 120000, "cumulative_loss_tokens": 22800000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 449.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.4, "frames": {"chat": 267}, "mem_gb": 16.11} +[eval step 190] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. Let's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n12 & -" +{"step": 191, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08201366777836035, "tokens": 120000, "cumulative_loss_tokens": 22920000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 248}, "mem_gb": 15.94} +{"step": 192, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.28501622741160293, "tokens": 120000, "cumulative_loss_tokens": 23040000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.979, "comp_len": 364.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.5, "frames": {"chat": 329}, "mem_gb": 15.93} +{"step": 193, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3122887461439706, "tokens": 120000, "cumulative_loss_tokens": 23160000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 354.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.2, "frames": {"chat": 339}, "mem_gb": 15.89} +{"step": 194, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.34632137398620444, "tokens": 120000, "cumulative_loss_tokens": 23280000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.4, "frames": {"chat": 354}, "mem_gb": 15.89} +{"step": 195, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1990443309733644, "tokens": 120000, "cumulative_loss_tokens": 23400000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.943, "comp_len": 430.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.6, "frames": {"chat": 279}, "mem_gb": 16.02} +{"step": 196, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24280436343842496, "tokens": 120000, "cumulative_loss_tokens": 23520000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.8, "frames": {"chat": 318}, "mem_gb": 15.95} +{"step": 197, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16376951620513575, "tokens": 120000, "cumulative_loss_tokens": 23640000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.875, "comp_len": 452.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.4, "frames": {"chat": 265}, "mem_gb": 16.04} +{"step": 198, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15522322897436097, "tokens": 120000, "cumulative_loss_tokens": 23760000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.85, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 254}, "mem_gb": 15.98} +{"step": 199, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23763825700944288, "tokens": 120000, "cumulative_loss_tokens": 23880000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.962, "comp_len": 381.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.0, "frames": {"chat": 315}, "mem_gb": 16.01} +{"step": 200, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.32632296637156977, "tokens": 120000, "cumulative_loss_tokens": 24000000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.4, "frames": {"chat": 300}, "mem_gb": 15.81} +[eval step 200] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. Let's break down the problem step-by-step:\n\n1. **Understand the Matrix:*" +{"step": 201, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21165221351847674, "tokens": 120000, "cumulative_loss_tokens": 24120000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.924, "comp_len": 434.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.8, "frames": {"chat": 276}, "mem_gb": 16.01} +{"step": 202, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.34457524228241915, "tokens": 120000, "cumulative_loss_tokens": 24240000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 337.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.4, "frames": {"chat": 356}, "mem_gb": 15.95} +{"step": 203, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.32126942253684004, "tokens": 120000, "cumulative_loss_tokens": 24360000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.9, "frames": {"chat": 304}, "mem_gb": 15.76} +{"step": 204, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.32354773480979104, "tokens": 120000, "cumulative_loss_tokens": 24480000, "grad_norm": 0.474609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 320.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.4, "frames": {"chat": 375}, "mem_gb": 15.68} +{"step": 205, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15825084536122158, "tokens": 120000, "cumulative_loss_tokens": 24600000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.939, "comp_len": 408.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.0, "frames": {"chat": 294}, "mem_gb": 16.04} +{"step": 206, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.29627488436199106, "tokens": 120000, "cumulative_loss_tokens": 24720000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 401.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.1, "frames": {"chat": 299}, "mem_gb": 16.05} +{"step": 207, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22168895639621963, "tokens": 120000, "cumulative_loss_tokens": 24840000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 393.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.6, "frames": {"chat": 305}, "mem_gb": 15.83} +{"step": 208, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1372577456595376, "tokens": 120000, "cumulative_loss_tokens": 24960000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.931, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 262}, "mem_gb": 15.88} +{"step": 209, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18872430870453827, "tokens": 120000, "cumulative_loss_tokens": 25080000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.95, "comp_len": 396.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.5, "frames": {"chat": 303}, "mem_gb": 15.94} +{"step": 210, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.32299288732521236, "tokens": 120000, "cumulative_loss_tokens": 25200000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 315.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.1, "frames": {"chat": 380}, "mem_gb": 15.82} +[eval step 210] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1' +{"step": 211, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15258993399993828, "tokens": 120000, "cumulative_loss_tokens": 25320000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.95, "comp_len": 425.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.5, "frames": {"chat": 282}, "mem_gb": 15.78} +{"step": 212, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2994012082218503, "tokens": 120000, "cumulative_loss_tokens": 25440000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 331.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.1, "frames": {"chat": 362}, "mem_gb": 15.79} +{"step": 213, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.29535049058605606, "tokens": 120000, "cumulative_loss_tokens": 25560000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.967, "comp_len": 390.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.1, "frames": {"chat": 307}, "mem_gb": 16.03} +{"step": 214, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18984129665071767, "tokens": 120000, "cumulative_loss_tokens": 25680000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 425.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.5, "frames": {"chat": 282}, "mem_gb": 15.9} +{"step": 215, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.31426690248246303, "tokens": 120000, "cumulative_loss_tokens": 25800000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.979, "comp_len": 356.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.7, "frames": {"chat": 337}, "mem_gb": 15.93} +{"step": 216, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18405989372702317, "tokens": 120000, "cumulative_loss_tokens": 25920000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.891, "comp_len": 436.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.5, "frames": {"chat": 275}, "mem_gb": 15.99} +{"step": 217, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3195844384454812, "tokens": 120000, "cumulative_loss_tokens": 26040000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.7, "frames": {"chat": 318}, "mem_gb": 15.8} +{"step": 218, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3616332326179681, "tokens": 120000, "cumulative_loss_tokens": 26160000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 311.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.0, "frames": {"chat": 385}, "mem_gb": 15.82} +{"step": 219, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3548837282889833, "tokens": 120000, "cumulative_loss_tokens": 26280000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 379.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.6, "frames": {"chat": 316}, "mem_gb": 15.78} +{"step": 220, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14040159034250924, "tokens": 120000, "cumulative_loss_tokens": 26400000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 451.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.2, "frames": {"chat": 266}, "mem_gb": 16.07} +[eval step 220] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly independe' +{"step": 221, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3375526880686482, "tokens": 120000, "cumulative_loss_tokens": 26520000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 323.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 71.4, "frames": {"chat": 371}, "mem_gb": 15.8} +{"step": 222, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16892142935271065, "tokens": 120000, "cumulative_loss_tokens": 26640000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.3, "frames": {"chat": 262}, "mem_gb": 16.01} +{"step": 223, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.34518754308577626, "tokens": 120000, "cumulative_loss_tokens": 26760000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 344.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.5, "frames": {"chat": 348}, "mem_gb": 15.83} +{"step": 224, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06776446208028744, "tokens": 120000, "cumulative_loss_tokens": 26880000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 222}, "mem_gb": 16.07} +{"step": 225, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2932193931110203, "tokens": 120000, "cumulative_loss_tokens": 27000000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.99, "comp_len": 294.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 71.5, "frames": {"chat": 408}, "mem_gb": 15.63} +{"step": 226, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2928274841371303, "tokens": 120000, "cumulative_loss_tokens": 27120000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 359.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.8, "frames": {"chat": 334}, "mem_gb": 15.7} +{"step": 227, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1546252207500084, "tokens": 120000, "cumulative_loss_tokens": 27240000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.968, "comp_len": 384.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.4, "frames": {"chat": 312}, "mem_gb": 15.75} +{"step": 228, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.29811865685375716, "tokens": 120000, "cumulative_loss_tokens": 27360000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 340.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.4, "frames": {"chat": 352}, "mem_gb": 15.73} +{"step": 229, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.261460732453006, "tokens": 120000, "cumulative_loss_tokens": 27480000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 323.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.9, "frames": {"chat": 371}, "mem_gb": 15.54} +{"step": 230, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17579592198502894, "tokens": 120000, "cumulative_loss_tokens": 27600000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.898, "comp_len": 422.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.7, "frames": {"chat": 284}, "mem_gb": 15.99} +[eval step 230] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as:\n\\[\nA = \\begin{bmatrix}\n12 & -16 & 4 " +{"step": 231, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15459353699831602, "tokens": 120000, "cumulative_loss_tokens": 27720000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 418.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 287}, "mem_gb": 16.05} +{"step": 232, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1228523612142851, "tokens": 120000, "cumulative_loss_tokens": 27840000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.916, "comp_len": 421.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 285}, "mem_gb": 15.95} +{"step": 233, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08205449555320665, "tokens": 120000, "cumulative_loss_tokens": 27960000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.873, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 252}, "mem_gb": 16.04} +{"step": 234, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2990638988738569, "tokens": 120000, "cumulative_loss_tokens": 28080000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 333.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.0, "frames": {"chat": 360}, "mem_gb": 15.83} +{"step": 235, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3220961626983713, "tokens": 120000, "cumulative_loss_tokens": 28200000, "grad_norm": 0.478515625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 334.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.7, "frames": {"chat": 359}, "mem_gb": 15.82} +{"step": 236, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3202049351937603, "tokens": 120000, "cumulative_loss_tokens": 28320000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 1.0, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.5, "frames": {"chat": 363}, "mem_gb": 15.58} +{"step": 237, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2816474720546665, "tokens": 120000, "cumulative_loss_tokens": 28440000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 332.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.2, "frames": {"chat": 361}, "mem_gb": 15.67} +{"step": 238, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1724256508340904, "tokens": 120000, "cumulative_loss_tokens": 28560000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.944, "comp_len": 419.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.5, "frames": {"chat": 286}, "mem_gb": 15.86} +{"step": 239, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30555949601754545, "tokens": 120000, "cumulative_loss_tokens": 28680000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 327.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 71.1, "frames": {"chat": 367}, "mem_gb": 15.71} +{"step": 240, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.32679198813159016, "tokens": 120000, "cumulative_loss_tokens": 28800000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 340.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.0, "frames": {"chat": 352}, "mem_gb": 15.74} +[eval step 240] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1' +{"step": 241, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18910810121611382, "tokens": 120000, "cumulative_loss_tokens": 28920000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.962, "comp_len": 382.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.0, "frames": {"chat": 314}, "mem_gb": 15.69} +{"step": 242, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2775511726691698, "tokens": 120000, "cumulative_loss_tokens": 29040000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.995, "comp_len": 322.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.9, "frames": {"chat": 372}, "mem_gb": 15.66} +{"step": 243, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2648515908466652, "tokens": 120000, "cumulative_loss_tokens": 29160000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 359.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.4, "frames": {"chat": 334}, "mem_gb": 15.81} +{"step": 244, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3331381598538719, "tokens": 120000, "cumulative_loss_tokens": 29280000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 326.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.7, "frames": {"chat": 368}, "mem_gb": 15.75} +{"step": 245, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.253562924684491, "tokens": 120000, "cumulative_loss_tokens": 29400000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.994, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.6, "frames": {"chat": 354}, "mem_gb": 15.9} +{"step": 246, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1559162168411538, "tokens": 120000, "cumulative_loss_tokens": 29520000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.973, "comp_len": 409.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.1, "frames": {"chat": 293}, "mem_gb": 15.79} +{"step": 247, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08934582893832897, "tokens": 120000, "cumulative_loss_tokens": 29640000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 231}, "mem_gb": 15.95} +{"step": 248, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22924559716898948, "tokens": 120000, "cumulative_loss_tokens": 29760000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.964, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.8, "frames": {"chat": 302}, "mem_gb": 15.92} +{"step": 249, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1503688322650579, "tokens": 120000, "cumulative_loss_tokens": 29880000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 456.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.0, "frames": {"chat": 263}, "mem_gb": 16.06} +{"step": 250, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16343837098814548, "tokens": 120000, "cumulative_loss_tokens": 30000000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 252}, "mem_gb": 15.99} +[eval step 250] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. Here's how we can do it step-by-step using Python and SymPy:\n\n1. **Defin" +{"step": 251, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2482833124493249, "tokens": 120000, "cumulative_loss_tokens": 30120000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.968, "comp_len": 348.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.3, "frames": {"chat": 344}, "mem_gb": 16.02} +{"step": 252, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23623603433938697, "tokens": 120000, "cumulative_loss_tokens": 30240000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.981, "comp_len": 383.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.7, "frames": {"chat": 313}, "mem_gb": 15.95} +{"step": 253, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.28484185952716506, "tokens": 120000, "cumulative_loss_tokens": 30360000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.992, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.1, "frames": {"chat": 358}, "mem_gb": 15.65} +{"step": 254, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14225151729526309, "tokens": 120000, "cumulative_loss_tokens": 30480000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.951, "comp_len": 449.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 267}, "mem_gb": 15.87} +{"step": 255, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2965879791254488, "tokens": 120000, "cumulative_loss_tokens": 30600000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.2, "frames": {"chat": 300}, "mem_gb": 15.74} +{"step": 256, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24699877228895203, "tokens": 120000, "cumulative_loss_tokens": 30720000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.994, "comp_len": 344.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.4, "frames": {"chat": 348}, "mem_gb": 15.68} +{"step": 257, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2815066012403152, "tokens": 120000, "cumulative_loss_tokens": 30840000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 364.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.1, "frames": {"chat": 329}, "mem_gb": 15.76} +{"step": 258, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.150311128562099, "tokens": 120000, "cumulative_loss_tokens": 30960000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.948, "comp_len": 446.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 269}, "mem_gb": 15.93} +{"step": 259, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2938167296140455, "tokens": 120000, "cumulative_loss_tokens": 31080000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.3, "frames": {"chat": 358}, "mem_gb": 15.6} +{"step": 260, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17735982676704104, "tokens": 120000, "cumulative_loss_tokens": 31200000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.96, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.2, "frames": {"chat": 302}, "mem_gb": 15.85} +[eval step 260] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1 &' +{"step": 261, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2994864141587478, "tokens": 120000, "cumulative_loss_tokens": 31320000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 315.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.0, "frames": {"chat": 381}, "mem_gb": 15.87} +{"step": 262, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3317916214861907, "tokens": 120000, "cumulative_loss_tokens": 31440000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 316.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 72.1, "frames": {"chat": 379}, "mem_gb": 15.82} +{"step": 263, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.32371730950316413, "tokens": 120000, "cumulative_loss_tokens": 31560000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 341.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.6, "frames": {"chat": 351}, "mem_gb": 15.64} +{"step": 264, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2773999908644706, "tokens": 120000, "cumulative_loss_tokens": 31680000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.3, "frames": {"chat": 330}, "mem_gb": 15.84} +{"step": 265, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2801876317330636, "tokens": 120000, "cumulative_loss_tokens": 31800000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.991, "comp_len": 379.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.8, "frames": {"chat": 316}, "mem_gb": 15.87} +{"step": 266, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19047235789267966, "tokens": 120000, "cumulative_loss_tokens": 31920000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.891, "comp_len": 434.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.3, "frames": {"chat": 276}, "mem_gb": 16.05} +{"step": 267, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17457375843231565, "tokens": 120000, "cumulative_loss_tokens": 32040000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 240}, "mem_gb": 15.99} +{"step": 268, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17561398149191712, "tokens": 120000, "cumulative_loss_tokens": 32160000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.948, "comp_len": 416.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.4, "frames": {"chat": 288}, "mem_gb": 15.9} +{"step": 269, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1686708824660629, "tokens": 120000, "cumulative_loss_tokens": 32280000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.937, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.0, "frames": {"chat": 271}, "mem_gb": 15.89} +{"step": 270, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.25845473671313374, "tokens": 120000, "cumulative_loss_tokens": 32400000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 336.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 70.7, "frames": {"chat": 357}, "mem_gb": 15.84} +[eval step 270] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given 4x4 matrix as:\n\n\\[\nA = \\begin{bmatrix}\n12 & " +{"step": 271, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3275356878739782, "tokens": 120000, "cumulative_loss_tokens": 32520000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.4, "frames": {"chat": 354}, "mem_gb": 15.81} +{"step": 272, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3089576366190488, "tokens": 120000, "cumulative_loss_tokens": 32640000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.927, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.8, "frames": {"chat": 302}, "mem_gb": 16.05} +{"step": 273, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23801056267358361, "tokens": 120000, "cumulative_loss_tokens": 32760000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 404.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.8, "frames": {"chat": 297}, "mem_gb": 16.04} +{"step": 274, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07905954725400856, "tokens": 120000, "cumulative_loss_tokens": 32880000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 219}, "mem_gb": 16.06} +{"step": 275, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3152970589374192, "tokens": 120000, "cumulative_loss_tokens": 33000000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.99, "comp_len": 392.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.3, "frames": {"chat": 306}, "mem_gb": 15.76} +{"step": 276, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3237693556587522, "tokens": 120000, "cumulative_loss_tokens": 33120000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 349.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.0, "frames": {"chat": 343}, "mem_gb": 15.7} +{"step": 277, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30347193134284267, "tokens": 120000, "cumulative_loss_tokens": 33240000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 351.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.0, "frames": {"chat": 341}, "mem_gb": 15.81} +{"step": 278, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20014061149423942, "tokens": 120000, "cumulative_loss_tokens": 33360000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.5, "frames": {"chat": 318}, "mem_gb": 15.76} +{"step": 279, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.31585417973902075, "tokens": 120000, "cumulative_loss_tokens": 33480000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 346.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.2, "frames": {"chat": 346}, "mem_gb": 15.65} +{"step": 280, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15319907906297595, "tokens": 120000, "cumulative_loss_tokens": 33600000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 433.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.0, "frames": {"chat": 277}, "mem_gb": 16.13} +[eval step 280] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's start by writing out the given matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16" +{"step": 281, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15190112631917, "tokens": 120000, "cumulative_loss_tokens": 33720000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.956, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.1, "frames": {"chat": 270}, "mem_gb": 15.86} +{"step": 282, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19110098611113305, "tokens": 120000, "cumulative_loss_tokens": 33840000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.6, "frames": {"chat": 257}, "mem_gb": 16.04} +{"step": 283, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30923019873052837, "tokens": 120000, "cumulative_loss_tokens": 33960000, "grad_norm": 0.478515625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.8, "frames": {"chat": 363}, "mem_gb": 15.65} +{"step": 284, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3194517805355291, "tokens": 120000, "cumulative_loss_tokens": 34080000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 337.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.2, "frames": {"chat": 356}, "mem_gb": 15.78} +{"step": 285, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.28514752473640254, "tokens": 120000, "cumulative_loss_tokens": 34200000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 371.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.7, "frames": {"chat": 323}, "mem_gb": 15.82} +{"step": 286, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14369000795896475, "tokens": 120000, "cumulative_loss_tokens": 34320000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 243}, "mem_gb": 16.07} +{"step": 287, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.288017972862348, "tokens": 120000, "cumulative_loss_tokens": 34440000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 390.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.2, "frames": {"chat": 307}, "mem_gb": 15.75} +{"step": 288, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.26883071097933375, "tokens": 120000, "cumulative_loss_tokens": 34560000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 346.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.5, "frames": {"chat": 346}, "mem_gb": 15.98} +{"step": 289, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17741179504264146, "tokens": 120000, "cumulative_loss_tokens": 34680000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.905, "comp_len": 439.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.2, "frames": {"chat": 273}, "mem_gb": 15.99} +{"step": 290, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19051366252149454, "tokens": 120000, "cumulative_loss_tokens": 34800000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.957, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.4, "frames": {"chat": 302}, "mem_gb": 15.92} +[eval step 290] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as:\n\\[\nA = \\begin{bmatrix}\n12 & -16 &" +{"step": 291, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2635435464471889, "tokens": 120000, "cumulative_loss_tokens": 34920000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.97, "comp_len": 356.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.9, "frames": {"chat": 337}, "mem_gb": 16.05} +{"step": 292, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20415386304327598, "tokens": 120000, "cumulative_loss_tokens": 35040000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 428.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.1, "frames": {"chat": 280}, "mem_gb": 15.92} +{"step": 293, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3529012825891686, "tokens": 120000, "cumulative_loss_tokens": 35160000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.7, "frames": {"chat": 331}, "mem_gb": 15.82} +{"step": 294, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20010651406816518, "tokens": 120000, "cumulative_loss_tokens": 35280000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 382.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.6, "frames": {"chat": 314}, "mem_gb": 15.93} +{"step": 295, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.348338770331877, "tokens": 120000, "cumulative_loss_tokens": 35400000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 408.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.6, "frames": {"chat": 294}, "mem_gb": 15.86} +{"step": 296, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16083920491451087, "tokens": 120000, "cumulative_loss_tokens": 35520000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.894, "comp_len": 438.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.9, "frames": {"chat": 274}, "mem_gb": 15.99} +{"step": 297, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2712820942190786, "tokens": 120000, "cumulative_loss_tokens": 35640000, "grad_norm": 0.58984375, "lr": 3e-05, "finish_rate": 0.995, "comp_len": 328.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.9, "frames": {"chat": 365}, "mem_gb": 15.71} +{"step": 298, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1760051498868658, "tokens": 120000, "cumulative_loss_tokens": 35760000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.972, "comp_len": 371.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.1, "frames": {"chat": 323}, "mem_gb": 15.77} +{"step": 299, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.125053847176147, "tokens": 120000, "cumulative_loss_tokens": 35880000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.829, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 234}, "mem_gb": 16.06} +{"step": 300, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16325586336916312, "tokens": 120000, "cumulative_loss_tokens": 36000000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.891, "comp_len": 449.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 267}, "mem_gb": 16.06} +[eval step 300] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as:\n\\[\nA = \\begin{bmatrix}\n12 & -16 &" +checkpoint snapshot queued -> outputs/healed/grid_general_fairness/reap_keep50_s1224_long500/step0300 +{"step": 301, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19214299598724272, "tokens": 120000, "cumulative_loss_tokens": 36120000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.961, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.1, "frames": {"chat": 304}, "mem_gb": 15.82} +{"step": 302, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.29611006540954116, "tokens": 120000, "cumulative_loss_tokens": 36240000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 321.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.3, "frames": {"chat": 373}, "mem_gb": 15.75} +{"step": 303, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17517327248199532, "tokens": 120000, "cumulative_loss_tokens": 36360000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.3, "frames": {"chat": 300}, "mem_gb": 16.03} +{"step": 304, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30671616884848724, "tokens": 120000, "cumulative_loss_tokens": 36480000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.982, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.8, "frames": {"chat": 331}, "mem_gb": 15.87} +{"step": 305, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1537431821726573, "tokens": 120000, "cumulative_loss_tokens": 36600000, "grad_norm": 1.1875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.1, "frames": {"chat": 259}, "mem_gb": 16.01} +{"step": 306, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2828215953375834, "tokens": 120000, "cumulative_loss_tokens": 36720000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.9, "frames": {"chat": 358}, "mem_gb": 15.63} +{"step": 307, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.31748190582487734, "tokens": 120000, "cumulative_loss_tokens": 36840000, "grad_norm": 0.4921875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.7, "frames": {"chat": 354}, "mem_gb": 15.88} +{"step": 308, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15749149380636712, "tokens": 120000, "cumulative_loss_tokens": 36960000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.939, "comp_len": 430.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.7, "frames": {"chat": 279}, "mem_gb": 15.95} +{"step": 309, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3186484920869271, "tokens": 120000, "cumulative_loss_tokens": 37080000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.0, "frames": {"chat": 354}, "mem_gb": 15.7} +{"step": 310, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.31272718894605833, "tokens": 120000, "cumulative_loss_tokens": 37200000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 359.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.1, "frames": {"chat": 334}, "mem_gb": 15.9} +[eval step 310] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -' +{"step": 311, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3153550029796548, "tokens": 120000, "cumulative_loss_tokens": 37320000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 367.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.1, "frames": {"chat": 327}, "mem_gb": 15.77} +{"step": 312, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.27661464767806854, "tokens": 120000, "cumulative_loss_tokens": 37440000, "grad_norm": 0.439453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 328.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.1, "frames": {"chat": 365}, "mem_gb": 15.87} +{"step": 313, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16841844340564371, "tokens": 120000, "cumulative_loss_tokens": 37560000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.869, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.6, "frames": {"chat": 252}, "mem_gb": 16.07} +{"step": 314, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14566575367416565, "tokens": 120000, "cumulative_loss_tokens": 37680000, "grad_norm": 0.54296875, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.5, "frames": {"chat": 256}, "mem_gb": 16.04} +{"step": 315, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1755091216189787, "tokens": 120000, "cumulative_loss_tokens": 37800000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.964, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 302}, "mem_gb": 15.81} +{"step": 316, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2780043101710267, "tokens": 120000, "cumulative_loss_tokens": 37920000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 342.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.0, "frames": {"chat": 350}, "mem_gb": 15.72} +{"step": 317, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17730869669849053, "tokens": 120000, "cumulative_loss_tokens": 38040000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 411.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.6, "frames": {"chat": 292}, "mem_gb": 15.84} +{"step": 318, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.29584056356226407, "tokens": 120000, "cumulative_loss_tokens": 38160000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 337.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.6, "frames": {"chat": 356}, "mem_gb": 15.65} +{"step": 319, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20022544026391892, "tokens": 120000, "cumulative_loss_tokens": 38280000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 345.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.2, "frames": {"chat": 347}, "mem_gb": 15.81} +{"step": 320, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18399523175458113, "tokens": 120000, "cumulative_loss_tokens": 38400000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.976, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.4, "frames": {"chat": 330}, "mem_gb": 16.01} +[eval step 320] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1 & ' +{"step": 321, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.29935896607286605, "tokens": 120000, "cumulative_loss_tokens": 38520000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 70.1, "frames": {"chat": 363}, "mem_gb": 15.87} +{"step": 322, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.27605618267795073, "tokens": 120000, "cumulative_loss_tokens": 38640000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.4, "frames": {"chat": 354}, "mem_gb": 15.73} +{"step": 323, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14932049081393828, "tokens": 120000, "cumulative_loss_tokens": 38760000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.93, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.2, "frames": {"chat": 270}, "mem_gb": 15.99} +{"step": 324, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2824427760565343, "tokens": 120000, "cumulative_loss_tokens": 38880000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 349.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.5, "frames": {"chat": 343}, "mem_gb": 15.77} +{"step": 325, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07517833311604336, "tokens": 120000, "cumulative_loss_tokens": 39000000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 232}, "mem_gb": 15.96} +{"step": 326, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.27200645623166736, "tokens": 120000, "cumulative_loss_tokens": 39120000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 346.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.4, "frames": {"chat": 346}, "mem_gb": 15.91} +{"step": 327, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21130462826990212, "tokens": 120000, "cumulative_loss_tokens": 39240000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.951, "comp_len": 392.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.5, "frames": {"chat": 306}, "mem_gb": 15.99} +{"step": 328, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14421452703930748, "tokens": 120000, "cumulative_loss_tokens": 39360000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 439.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 273}, "mem_gb": 16.04} +{"step": 329, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1527492280335476, "tokens": 120000, "cumulative_loss_tokens": 39480000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 451.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.3, "frames": {"chat": 266}, "mem_gb": 15.86} +{"step": 330, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2298182229324865, "tokens": 120000, "cumulative_loss_tokens": 39600000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.936, "comp_len": 401.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.2, "frames": {"chat": 299}, "mem_gb": 16.06} +[eval step 330] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1' +{"step": 331, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15647198322201147, "tokens": 120000, "cumulative_loss_tokens": 39720000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.88, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 259}, "mem_gb": 16.14} +{"step": 332, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19848604813426113, "tokens": 120000, "cumulative_loss_tokens": 39840000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.966, "comp_len": 369.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.9, "frames": {"chat": 325}, "mem_gb": 15.88} +{"step": 333, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19916361242351122, "tokens": 120000, "cumulative_loss_tokens": 39960000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.9, "frames": {"chat": 271}, "mem_gb": 15.95} +{"step": 334, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2742750384826213, "tokens": 120000, "cumulative_loss_tokens": 40080000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 334.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.7, "frames": {"chat": 359}, "mem_gb": 15.74} +{"step": 335, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2677033269007845, "tokens": 120000, "cumulative_loss_tokens": 40200000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 347.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.9, "frames": {"chat": 345}, "mem_gb": 15.8} +{"step": 336, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30258281591987857, "tokens": 120000, "cumulative_loss_tokens": 40320000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 301.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.2, "frames": {"chat": 398}, "mem_gb": 15.83} +{"step": 337, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1772353317519029, "tokens": 120000, "cumulative_loss_tokens": 40440000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.964, "comp_len": 390.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.7, "frames": {"chat": 307}, "mem_gb": 16.03} +{"step": 338, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.29415553405807976, "tokens": 120000, "cumulative_loss_tokens": 40560000, "grad_norm": 0.478515625, "lr": 3e-05, "finish_rate": 0.995, "comp_len": 308.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 72.6, "frames": {"chat": 389}, "mem_gb": 15.8} +{"step": 339, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05377294343154256, "tokens": 120000, "cumulative_loss_tokens": 40680000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 231}, "mem_gb": 16.04} +{"step": 340, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2708478927415485, "tokens": 120000, "cumulative_loss_tokens": 40800000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 373.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.7, "frames": {"chat": 321}, "mem_gb": 15.71} +[eval step 340] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly independent rows or col' +{"step": 341, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3198238567600027, "tokens": 120000, "cumulative_loss_tokens": 40920000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 357.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.2, "frames": {"chat": 336}, "mem_gb": 15.73} +{"step": 342, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2852054091291968, "tokens": 120000, "cumulative_loss_tokens": 41040000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 355.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.3, "frames": {"chat": 338}, "mem_gb": 15.89} +{"step": 343, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16390295645698594, "tokens": 120000, "cumulative_loss_tokens": 41160000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.945, "comp_len": 436.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 275}, "mem_gb": 16.03} +{"step": 344, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11328129321429878, "tokens": 120000, "cumulative_loss_tokens": 41280000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 226}, "mem_gb": 16.05} +{"step": 345, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3046031262753221, "tokens": 120000, "cumulative_loss_tokens": 41400000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 346.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.0, "frames": {"chat": 346}, "mem_gb": 15.85} +{"step": 346, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2956097679840401, "tokens": 120000, "cumulative_loss_tokens": 41520000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.8, "frames": {"chat": 363}, "mem_gb": 15.66} +{"step": 347, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16160289938263595, "tokens": 120000, "cumulative_loss_tokens": 41640000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.955, "comp_len": 383.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.9, "frames": {"chat": 313}, "mem_gb": 15.76} +{"step": 348, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.32252234573277333, "tokens": 120000, "cumulative_loss_tokens": 41760000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 323.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.0, "frames": {"chat": 371}, "mem_gb": 15.82} +{"step": 349, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23571122382180765, "tokens": 120000, "cumulative_loss_tokens": 41880000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 373.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.8, "frames": {"chat": 321}, "mem_gb": 15.6} +{"step": 350, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.28336655547525735, "tokens": 120000, "cumulative_loss_tokens": 42000000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 365.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.3, "frames": {"chat": 328}, "mem_gb": 15.68} +[eval step 350] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1 &' +{"step": 351, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3651294656487182, "tokens": 120000, "cumulative_loss_tokens": 42120000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 357.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.4, "frames": {"chat": 336}, "mem_gb": 15.9} +{"step": 352, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.292059118018408, "tokens": 120000, "cumulative_loss_tokens": 42240000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 323.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.6, "frames": {"chat": 371}, "mem_gb": 15.79} +{"step": 353, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07931742044407875, "tokens": 120000, "cumulative_loss_tokens": 42360000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 226}, "mem_gb": 16.03} +{"step": 354, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2550304350883079, "tokens": 120000, "cumulative_loss_tokens": 42480000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.931, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.0, "frames": {"chat": 304}, "mem_gb": 16.04} +{"step": 355, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07330653709719578, "tokens": 120000, "cumulative_loss_tokens": 42600000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.5, "frames": {"chat": 217}, "mem_gb": 16.03} +{"step": 356, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.25114847851675004, "tokens": 120000, "cumulative_loss_tokens": 42720000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 351.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.6, "frames": {"chat": 341}, "mem_gb": 15.72} +{"step": 357, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2891262057554598, "tokens": 120000, "cumulative_loss_tokens": 42840000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 343.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.3, "frames": {"chat": 349}, "mem_gb": 15.62} +{"step": 358, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.31588473438881337, "tokens": 120000, "cumulative_loss_tokens": 42960000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 336.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 70.5, "frames": {"chat": 357}, "mem_gb": 15.82} +{"step": 359, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.33899677562055486, "tokens": 120000, "cumulative_loss_tokens": 43080000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 325.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.9, "frames": {"chat": 369}, "mem_gb": 15.6} +{"step": 360, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17974190369814944, "tokens": 120000, "cumulative_loss_tokens": 43200000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.974, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.2, "frames": {"chat": 302}, "mem_gb": 15.88} +[eval step 360] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as:\n\n\\[\nA = \\begin{bmatrix}\n12 & -16 " +{"step": 361, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17992556028673426, "tokens": 120000, "cumulative_loss_tokens": 43320000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.968, "comp_len": 385.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.6, "frames": {"chat": 311}, "mem_gb": 15.69} +{"step": 362, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3073442754408345, "tokens": 120000, "cumulative_loss_tokens": 43440000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 365.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.9, "frames": {"chat": 328}, "mem_gb": 15.78} +{"step": 363, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3553521730494375, "tokens": 120000, "cumulative_loss_tokens": 43560000, "grad_norm": 0.546875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.6, "frames": {"chat": 353}, "mem_gb": 15.82} +{"step": 364, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2637005336632642, "tokens": 120000, "cumulative_loss_tokens": 43680000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 360.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.5, "frames": {"chat": 333}, "mem_gb": 15.89} +{"step": 365, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19176016737166793, "tokens": 120000, "cumulative_loss_tokens": 43800000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.966, "comp_len": 375.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.9, "frames": {"chat": 320}, "mem_gb": 15.62} +{"step": 366, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09709547475147992, "tokens": 120000, "cumulative_loss_tokens": 43920000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 217}, "mem_gb": 16.13} +{"step": 367, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.060408217743970455, "tokens": 120000, "cumulative_loss_tokens": 44040000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.849, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 219}, "mem_gb": 16.05} +{"step": 368, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19196979975022066, "tokens": 120000, "cumulative_loss_tokens": 44160000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.905, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 262}, "mem_gb": 15.96} +{"step": 369, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3134816191433153, "tokens": 120000, "cumulative_loss_tokens": 44280000, "grad_norm": 0.53125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 320.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 70.1, "frames": {"chat": 375}, "mem_gb": 15.85} +{"step": 370, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.26572082690556226, "tokens": 120000, "cumulative_loss_tokens": 44400000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.8, "frames": {"chat": 342}, "mem_gb": 15.7} +[eval step 370] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as:\n\n\\[\nA = \\begin{bmatrix}\n12 & -16 " +{"step": 371, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1550645599910058, "tokens": 120000, "cumulative_loss_tokens": 44520000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.945, "comp_len": 389.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.5, "frames": {"chat": 308}, "mem_gb": 16.01} +{"step": 372, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0440521934287933, "tokens": 120000, "cumulative_loss_tokens": 44640000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 211}, "mem_gb": 16.04} +{"step": 373, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11794430323278841, "tokens": 120000, "cumulative_loss_tokens": 44760000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 427.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.8, "frames": {"chat": 281}, "mem_gb": 16.04} +{"step": 374, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24780444338647648, "tokens": 120000, "cumulative_loss_tokens": 44880000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 327.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.2, "frames": {"chat": 367}, "mem_gb": 15.59} +{"step": 375, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24827632907290634, "tokens": 120000, "cumulative_loss_tokens": 45000000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 384.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.5, "frames": {"chat": 312}, "mem_gb": 15.75} +{"step": 376, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18168822836314016, "tokens": 120000, "cumulative_loss_tokens": 45120000, "grad_norm": 0.48828125, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.3, "frames": {"chat": 271}, "mem_gb": 16.06} +{"step": 377, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11967306855365945, "tokens": 120000, "cumulative_loss_tokens": 45240000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 446.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.6, "frames": {"chat": 269}, "mem_gb": 16.03} +{"step": 378, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.23304759976395095, "tokens": 120000, "cumulative_loss_tokens": 45360000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.989, "comp_len": 340.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.9, "frames": {"chat": 352}, "mem_gb": 15.55} +{"step": 379, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10071651058743397, "tokens": 120000, "cumulative_loss_tokens": 45480000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 243}, "mem_gb": 16.04} +{"step": 380, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15536279593119398, "tokens": 120000, "cumulative_loss_tokens": 45600000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.952, "comp_len": 409.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.3, "frames": {"chat": 293}, "mem_gb": 15.96} +[eval step 380] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as:\n\\[\nA = \\begin{bmatrix}\n12 & -16 &" +{"step": 381, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08492357026590035, "tokens": 120000, "cumulative_loss_tokens": 45720000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.872, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.3, "frames": {"chat": 242}, "mem_gb": 16.04} +{"step": 382, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.25294500040340545, "tokens": 120000, "cumulative_loss_tokens": 45840000, "grad_norm": 0.439453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.1, "frames": {"chat": 353}, "mem_gb": 15.73} +{"step": 383, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2325842794330713, "tokens": 120000, "cumulative_loss_tokens": 45960000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 355.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.4, "frames": {"chat": 338}, "mem_gb": 15.74} +{"step": 384, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12266503979844662, "tokens": 120000, "cumulative_loss_tokens": 46080000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.7, "frames": {"chat": 259}, "mem_gb": 16.04} +{"step": 385, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.22742056115831558, "tokens": 120000, "cumulative_loss_tokens": 46200000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 311.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.9, "frames": {"chat": 385}, "mem_gb": 15.59} +{"step": 386, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20938693722110863, "tokens": 120000, "cumulative_loss_tokens": 46320000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 338.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.4, "frames": {"chat": 355}, "mem_gb": 15.75} +{"step": 387, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20082530200366552, "tokens": 120000, "cumulative_loss_tokens": 46440000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 365.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.7, "frames": {"chat": 328}, "mem_gb": 15.66} +{"step": 388, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21147014240783948, "tokens": 120000, "cumulative_loss_tokens": 46560000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 338.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.6, "frames": {"chat": 355}, "mem_gb": 15.81} +{"step": 389, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.25787362239655726, "tokens": 120000, "cumulative_loss_tokens": 46680000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 327.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.2, "frames": {"chat": 367}, "mem_gb": 15.68} +{"step": 390, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.22498693597537156, "tokens": 120000, "cumulative_loss_tokens": 46800000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 349.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.4, "frames": {"chat": 343}, "mem_gb": 15.72} +[eval step 390] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as:\n\n\\[\nA = \\begin{bmatrix}\n12 & -16 " +{"step": 391, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05578648099261336, "tokens": 120000, "cumulative_loss_tokens": 46920000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 244}, "mem_gb": 16.04} +{"step": 392, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.138854704477638, "tokens": 120000, "cumulative_loss_tokens": 47040000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 354.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.5, "frames": {"chat": 339}, "mem_gb": 15.88} +{"step": 393, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1433349025327266, "tokens": 120000, "cumulative_loss_tokens": 47160000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 402.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.4, "frames": {"chat": 298}, "mem_gb": 16.03} +{"step": 394, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2227652705781317, "tokens": 120000, "cumulative_loss_tokens": 47280000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 301.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 72.3, "frames": {"chat": 398}, "mem_gb": 15.9} +{"step": 395, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21596198100456968, "tokens": 120000, "cumulative_loss_tokens": 47400000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 331.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.7, "frames": {"chat": 362}, "mem_gb": 16.01} +{"step": 396, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14203152840395147, "tokens": 120000, "cumulative_loss_tokens": 47520000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 383.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.9, "frames": {"chat": 313}, "mem_gb": 15.88} +{"step": 397, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2535869244144919, "tokens": 120000, "cumulative_loss_tokens": 47640000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 370.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.3, "frames": {"chat": 324}, "mem_gb": 15.79} +{"step": 398, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.154214827561751, "tokens": 120000, "cumulative_loss_tokens": 47760000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.936, "comp_len": 382.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.3, "frames": {"chat": 314}, "mem_gb": 15.85} +{"step": 399, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.28178424551251036, "tokens": 120000, "cumulative_loss_tokens": 47880000, "grad_norm": 0.546875, "lr": 3e-05, "finish_rate": 0.991, "comp_len": 347.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.8, "frames": {"chat": 345}, "mem_gb": 15.77} +{"step": 400, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11000088793364508, "tokens": 120000, "cumulative_loss_tokens": 48000000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.897, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 262}, "mem_gb": 15.95} +[eval step 400] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1 & ' +{"step": 401, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2596910029921681, "tokens": 120000, "cumulative_loss_tokens": 48120000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 321.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.1, "frames": {"chat": 373}, "mem_gb": 15.74} +{"step": 402, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2233935168984967, "tokens": 120000, "cumulative_loss_tokens": 48240000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 359.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.8, "frames": {"chat": 334}, "mem_gb": 15.74} +{"step": 403, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16129895362847796, "tokens": 120000, "cumulative_loss_tokens": 48360000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 367.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.3, "frames": {"chat": 327}, "mem_gb": 15.89} +{"step": 404, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.040422961515731486, "tokens": 120000, "cumulative_loss_tokens": 48480000, "grad_norm": 0.228515625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 238}, "mem_gb": 16.06} +{"step": 405, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18921210524314083, "tokens": 120000, "cumulative_loss_tokens": 48600000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.961, "comp_len": 387.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.6, "frames": {"chat": 310}, "mem_gb": 15.79} +{"step": 406, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.22731006532437167, "tokens": 120000, "cumulative_loss_tokens": 48720000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 345.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.2, "frames": {"chat": 347}, "mem_gb": 15.79} +{"step": 407, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2263931953029707, "tokens": 120000, "cumulative_loss_tokens": 48840000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.4, "frames": {"chat": 304}, "mem_gb": 15.72} +{"step": 408, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21199652055079737, "tokens": 120000, "cumulative_loss_tokens": 48960000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 349.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.4, "frames": {"chat": 343}, "mem_gb": 15.74} +{"step": 409, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24227636873464412, "tokens": 120000, "cumulative_loss_tokens": 49080000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 382.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.5, "frames": {"chat": 314}, "mem_gb": 15.73} +{"step": 410, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2166262756802452, "tokens": 120000, "cumulative_loss_tokens": 49200000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.97, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.9, "frames": {"chat": 300}, "mem_gb": 15.83} +[eval step 410] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1' +{"step": 411, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17053968869245922, "tokens": 120000, "cumulative_loss_tokens": 49320000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.981, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.2, "frames": {"chat": 318}, "mem_gb": 15.88} +{"step": 412, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12892872311325435, "tokens": 120000, "cumulative_loss_tokens": 49440000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.939, "comp_len": 430.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.7, "frames": {"chat": 279}, "mem_gb": 15.91} +{"step": 413, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.22957993539332722, "tokens": 120000, "cumulative_loss_tokens": 49560000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.984, "comp_len": 329.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.0, "frames": {"chat": 364}, "mem_gb": 15.98} +{"step": 414, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15994093478830376, "tokens": 120000, "cumulative_loss_tokens": 49680000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 256}, "mem_gb": 16.04} +{"step": 415, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.282432095315028, "tokens": 120000, "cumulative_loss_tokens": 49800000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 352.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.4, "frames": {"chat": 340}, "mem_gb": 15.75} +{"step": 416, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1221779058281177, "tokens": 120000, "cumulative_loss_tokens": 49920000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.901, "comp_len": 424.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.1, "frames": {"chat": 283}, "mem_gb": 15.97} +{"step": 417, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.23161045495119567, "tokens": 120000, "cumulative_loss_tokens": 50040000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 343.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.6, "frames": {"chat": 349}, "mem_gb": 15.76} +{"step": 418, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12766366917919367, "tokens": 120000, "cumulative_loss_tokens": 50160000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.963, "comp_len": 404.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.6, "frames": {"chat": 297}, "mem_gb": 15.83} +{"step": 419, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.28232319180596777, "tokens": 120000, "cumulative_loss_tokens": 50280000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 334.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.4, "frames": {"chat": 359}, "mem_gb": 15.8} +{"step": 420, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17352004654083092, "tokens": 120000, "cumulative_loss_tokens": 50400000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.5, "frames": {"chat": 342}, "mem_gb": 15.85} +[eval step 420] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as:\n\\[\nA = \\begin{bmatrix}\n12 & -16 &" +{"step": 421, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.115978686744549, "tokens": 120000, "cumulative_loss_tokens": 50520000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 446.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 269}, "mem_gb": 16.01} +{"step": 422, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16951592262610793, "tokens": 120000, "cumulative_loss_tokens": 50640000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.963, "comp_len": 370.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.9, "frames": {"chat": 324}, "mem_gb": 16.0} +{"step": 423, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1323940156360312, "tokens": 120000, "cumulative_loss_tokens": 50760000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.94, "comp_len": 427.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.1, "frames": {"chat": 281}, "mem_gb": 15.99} +{"step": 424, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1660741152546679, "tokens": 120000, "cumulative_loss_tokens": 50880000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.916, "comp_len": 418.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.7, "frames": {"chat": 287}, "mem_gb": 16.04} +{"step": 425, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1969115988909267, "tokens": 120000, "cumulative_loss_tokens": 51000000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.939, "comp_len": 405.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.7, "frames": {"chat": 296}, "mem_gb": 16.05} +{"step": 426, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.135201798595783, "tokens": 120000, "cumulative_loss_tokens": 51120000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 449.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.2, "frames": {"chat": 267}, "mem_gb": 16.1} +{"step": 427, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13638714981228114, "tokens": 120000, "cumulative_loss_tokens": 51240000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.2, "frames": {"chat": 249}, "mem_gb": 16.0} +{"step": 428, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2490285068643706, "tokens": 120000, "cumulative_loss_tokens": 51360000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.3, "frames": {"chat": 363}, "mem_gb": 15.7} +{"step": 429, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1621854290435401, "tokens": 120000, "cumulative_loss_tokens": 51480000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.953, "comp_len": 401.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.7, "frames": {"chat": 299}, "mem_gb": 15.83} +{"step": 430, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24157769010777896, "tokens": 120000, "cumulative_loss_tokens": 51600000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.2, "frames": {"chat": 331}, "mem_gb": 15.71} +[eval step 430] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as:\n\\[\nA = \\begin{bmatrix}\n12 & -16 &" +{"step": 431, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12677099231950317, "tokens": 120000, "cumulative_loss_tokens": 51720000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.954, "comp_len": 428.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 280}, "mem_gb": 15.92} +{"step": 432, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24915074762748554, "tokens": 120000, "cumulative_loss_tokens": 51840000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.994, "comp_len": 342.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.3, "frames": {"chat": 350}, "mem_gb": 15.91} +{"step": 433, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.23871512211148316, "tokens": 120000, "cumulative_loss_tokens": 51960000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 373.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.0, "frames": {"chat": 321}, "mem_gb": 15.67} +{"step": 434, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12178932635299862, "tokens": 120000, "cumulative_loss_tokens": 52080000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.943, "comp_len": 427.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.6, "frames": {"chat": 281}, "mem_gb": 15.97} +{"step": 435, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.275969661668269, "tokens": 120000, "cumulative_loss_tokens": 52200000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.1, "frames": {"chat": 331}, "mem_gb": 15.65} +{"step": 436, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2017137859325235, "tokens": 120000, "cumulative_loss_tokens": 52320000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.956, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.5, "frames": {"chat": 318}, "mem_gb": 16.02} +{"step": 437, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15842565876391407, "tokens": 120000, "cumulative_loss_tokens": 52440000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.942, "comp_len": 388.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.4, "frames": {"chat": 309}, "mem_gb": 16.04} +{"step": 438, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.152509997984441, "tokens": 120000, "cumulative_loss_tokens": 52560000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.941, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.4, "frames": {"chat": 304}, "mem_gb": 15.93} +{"step": 439, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10116779167731292, "tokens": 120000, "cumulative_loss_tokens": 52680000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 430.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 279}, "mem_gb": 15.95} +{"step": 440, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04737602149757246, "tokens": 120000, "cumulative_loss_tokens": 52800000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 248}, "mem_gb": 15.95} +[eval step 440] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n1" +{"step": 441, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18501704499078914, "tokens": 120000, "cumulative_loss_tokens": 52920000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.907, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.2, "frames": {"chat": 268}, "mem_gb": 15.92} +{"step": 442, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18275448781829326, "tokens": 120000, "cumulative_loss_tokens": 53040000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.953, "comp_len": 401.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.3, "frames": {"chat": 299}, "mem_gb": 16.02} +{"step": 443, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2030420279776988, "tokens": 120000, "cumulative_loss_tokens": 53160000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.978, "comp_len": 372.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.0, "frames": {"chat": 322}, "mem_gb": 15.88} +{"step": 444, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11846702408028456, "tokens": 120000, "cumulative_loss_tokens": 53280000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.8, "frames": {"chat": 248}, "mem_gb": 16.06} +{"step": 445, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18047660849990013, "tokens": 120000, "cumulative_loss_tokens": 53400000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 368.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.1, "frames": {"chat": 326}, "mem_gb": 15.95} +{"step": 446, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1636732902829225, "tokens": 120000, "cumulative_loss_tokens": 53520000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.984, "comp_len": 382.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.7, "frames": {"chat": 314}, "mem_gb": 15.71} +{"step": 447, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04606743650698724, "tokens": 120000, "cumulative_loss_tokens": 53640000, "grad_norm": 0.24609375, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 232}, "mem_gb": 16.05} +{"step": 448, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24397644081069156, "tokens": 120000, "cumulative_loss_tokens": 53760000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.994, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.2, "frames": {"chat": 330}, "mem_gb": 15.78} +{"step": 449, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1230831551321627, "tokens": 120000, "cumulative_loss_tokens": 53880000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.901, "comp_len": 411.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.2, "frames": {"chat": 292}, "mem_gb": 15.93} +{"step": 450, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21141370300810475, "tokens": 120000, "cumulative_loss_tokens": 54000000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.978, "comp_len": 384.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.8, "frames": {"chat": 312}, "mem_gb": 15.95} +[eval step 450] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n1" +checkpoint snapshot queued -> outputs/healed/grid_general_fairness/reap_keep50_s1224_long500/step0450 +{"step": 451, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.25736966460969607, "tokens": 120000, "cumulative_loss_tokens": 54120000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 314.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 71.4, "frames": {"chat": 382}, "mem_gb": 15.76} +{"step": 452, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2673872146341484, "tokens": 120000, "cumulative_loss_tokens": 54240000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.994, "comp_len": 333.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 67.7, "frames": {"chat": 360}, "mem_gb": 15.63} +{"step": 453, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03807283711478424, "tokens": 120000, "cumulative_loss_tokens": 54360000, "grad_norm": 0.2314453125, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 222}, "mem_gb": 15.89} +{"step": 454, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24587370012854226, "tokens": 120000, "cumulative_loss_tokens": 54480000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.7, "frames": {"chat": 342}, "mem_gb": 15.94} +{"step": 455, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.23798172148214652, "tokens": 120000, "cumulative_loss_tokens": 54600000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 356.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.0, "frames": {"chat": 337}, "mem_gb": 15.74} +{"step": 456, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15561251825699582, "tokens": 120000, "cumulative_loss_tokens": 54720000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.934, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.2, "frames": {"chat": 304}, "mem_gb": 15.99} +{"step": 457, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16316290256114055, "tokens": 120000, "cumulative_loss_tokens": 54840000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.954, "comp_len": 368.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.9, "frames": {"chat": 326}, "mem_gb": 15.89} +{"step": 458, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20806148803344307, "tokens": 120000, "cumulative_loss_tokens": 54960000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 387.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.8, "frames": {"chat": 310}, "mem_gb": 15.95} +{"step": 459, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2053287281407509, "tokens": 120000, "cumulative_loss_tokens": 55080000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.982, "comp_len": 365.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.0, "frames": {"chat": 328}, "mem_gb": 15.77} +{"step": 460, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24368232978830734, "tokens": 120000, "cumulative_loss_tokens": 55200000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 345.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.2, "frames": {"chat": 347}, "mem_gb": 15.6} +[eval step 460] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1' +{"step": 461, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19848121287118023, "tokens": 120000, "cumulative_loss_tokens": 55320000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.976, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.9, "frames": {"chat": 330}, "mem_gb": 15.89} +{"step": 462, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0501603814638375, "tokens": 120000, "cumulative_loss_tokens": 55440000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.751, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 209}, "mem_gb": 16.09} +{"step": 463, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.23107828773936878, "tokens": 120000, "cumulative_loss_tokens": 55560000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.99, "comp_len": 300.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 71.2, "frames": {"chat": 399}, "mem_gb": 15.67} +{"step": 464, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20115915480243662, "tokens": 120000, "cumulative_loss_tokens": 55680000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 341.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.9, "frames": {"chat": 351}, "mem_gb": 15.75} +{"step": 465, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15638394095649322, "tokens": 120000, "cumulative_loss_tokens": 55800000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.975, "comp_len": 372.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 60.0, "frames": {"chat": 322}, "mem_gb": 15.75} +{"step": 466, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.23955250265694533, "tokens": 120000, "cumulative_loss_tokens": 55920000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 360.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.7, "frames": {"chat": 333}, "mem_gb": 15.64} +{"step": 467, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17247664903791932, "tokens": 120000, "cumulative_loss_tokens": 56040000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.955, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.8, "frames": {"chat": 330}, "mem_gb": 15.93} +{"step": 468, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.038492249800621846, "tokens": 120000, "cumulative_loss_tokens": 56160000, "grad_norm": 0.2255859375, "lr": 3e-05, "finish_rate": 0.866, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 238}, "mem_gb": 15.96} +{"step": 469, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10537788205655912, "tokens": 120000, "cumulative_loss_tokens": 56280000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.943, "comp_len": 404.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.9, "frames": {"chat": 297}, "mem_gb": 15.87} +{"step": 470, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21190769448286542, "tokens": 120000, "cumulative_loss_tokens": 56400000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.991, "comp_len": 347.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.4, "frames": {"chat": 345}, "mem_gb": 15.83} +[eval step 470] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n1" +{"step": 471, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1485939651843471, "tokens": 120000, "cumulative_loss_tokens": 56520000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.944, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.9, "frames": {"chat": 304}, "mem_gb": 15.98} +{"step": 472, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20662977668990692, "tokens": 120000, "cumulative_loss_tokens": 56640000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.982, "comp_len": 368.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.8, "frames": {"chat": 326}, "mem_gb": 15.98} +{"step": 473, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24409980630020922, "tokens": 120000, "cumulative_loss_tokens": 56760000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 345.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.6, "frames": {"chat": 347}, "mem_gb": 15.91} +{"step": 474, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2306728854667085, "tokens": 120000, "cumulative_loss_tokens": 56880000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 369.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.2, "frames": {"chat": 325}, "mem_gb": 15.79} +{"step": 475, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1324356382307286, "tokens": 120000, "cumulative_loss_tokens": 57000000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.933, "comp_len": 401.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.5, "frames": {"chat": 299}, "mem_gb": 15.98} +{"step": 476, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2680205172271623, "tokens": 120000, "cumulative_loss_tokens": 57120000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 347.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.1, "frames": {"chat": 345}, "mem_gb": 15.75} +{"step": 477, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2335554752687613, "tokens": 120000, "cumulative_loss_tokens": 57240000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 348.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.0, "frames": {"chat": 344}, "mem_gb": 15.65} +{"step": 478, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17294441227380497, "tokens": 120000, "cumulative_loss_tokens": 57360000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.985, "comp_len": 348.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.3, "frames": {"chat": 344}, "mem_gb": 15.91} +{"step": 479, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.161349027118211, "tokens": 120000, "cumulative_loss_tokens": 57480000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.962, "comp_len": 416.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.3, "frames": {"chat": 288}, "mem_gb": 15.98} +{"step": 480, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15392080484014004, "tokens": 120000, "cumulative_loss_tokens": 57600000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 396.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.9, "frames": {"chat": 303}, "mem_gb": 15.68} +[eval step 480] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1' +{"step": 481, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10330657383169357, "tokens": 120000, "cumulative_loss_tokens": 57720000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.898, "comp_len": 422.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.6, "frames": {"chat": 284}, "mem_gb": 15.87} +{"step": 482, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.25846441519111396, "tokens": 120000, "cumulative_loss_tokens": 57840000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 325.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.5, "frames": {"chat": 369}, "mem_gb": 15.76} +{"step": 483, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1187284121698079, "tokens": 120000, "cumulative_loss_tokens": 57960000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 446.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 269}, "mem_gb": 16.02} +{"step": 484, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.25894636472339433, "tokens": 120000, "cumulative_loss_tokens": 58080000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 351.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.7, "frames": {"chat": 341}, "mem_gb": 15.78} +{"step": 485, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24258800733716537, "tokens": 120000, "cumulative_loss_tokens": 58200000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 312.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.5, "frames": {"chat": 384}, "mem_gb": 15.55} +{"step": 486, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21789499796386808, "tokens": 120000, "cumulative_loss_tokens": 58320000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.6, "frames": {"chat": 358}, "mem_gb": 15.71} +{"step": 487, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.22395840081153437, "tokens": 120000, "cumulative_loss_tokens": 58440000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.9, "frames": {"chat": 342}, "mem_gb": 15.89} +{"step": 488, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2561194989992616, "tokens": 120000, "cumulative_loss_tokens": 58560000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 328.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.3, "frames": {"chat": 365}, "mem_gb": 15.71} +{"step": 489, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.23198317301028099, "tokens": 120000, "cumulative_loss_tokens": 58680000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 356.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.4, "frames": {"chat": 337}, "mem_gb": 15.94} +{"step": 490, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20493629304049538, "tokens": 120000, "cumulative_loss_tokens": 58800000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.961, "comp_len": 360.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.9, "frames": {"chat": 333}, "mem_gb": 16.04} +[eval step 490] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. Here's how we can do it step-by-step using Python and the `numpy` librar" +{"step": 491, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14055722026024015, "tokens": 120000, "cumulative_loss_tokens": 58920000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 409.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.7, "frames": {"chat": 293}, "mem_gb": 16.08} +{"step": 492, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12293431687603394, "tokens": 120000, "cumulative_loss_tokens": 59040000, "grad_norm": 0.298828125, "lr": 3e-05, "finish_rate": 0.973, "comp_len": 412.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.3, "frames": {"chat": 291}, "mem_gb": 15.84} +{"step": 493, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20858711904548108, "tokens": 120000, "cumulative_loss_tokens": 59160000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 351.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 66.0, "frames": {"chat": 341}, "mem_gb": 15.87} +{"step": 494, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13247924134900482, "tokens": 120000, "cumulative_loss_tokens": 59280000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.97, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.4, "frames": {"chat": 300}, "mem_gb": 15.81} +{"step": 495, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11377064473503269, "tokens": 120000, "cumulative_loss_tokens": 59400000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.885, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.9, "frames": {"chat": 270}, "mem_gb": 16.03} +{"step": 496, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09712668888554908, "tokens": 120000, "cumulative_loss_tokens": 59520000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 259}, "mem_gb": 16.05} +{"step": 497, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.23318171648929517, "tokens": 120000, "cumulative_loss_tokens": 59640000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 338.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 65.0, "frames": {"chat": 355}, "mem_gb": 15.62} +{"step": 498, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.25001415310632435, "tokens": 120000, "cumulative_loss_tokens": 59760000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 309.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 69.2, "frames": {"chat": 388}, "mem_gb": 15.98} +{"step": 499, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21438722058841958, "tokens": 120000, "cumulative_loss_tokens": 59880000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 61.3, "frames": {"chat": 318}, "mem_gb": 15.7} +{"step": 500, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24798546382077039, "tokens": 120000, "cumulative_loss_tokens": 60000000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 313.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 68.9, "frames": {"chat": 383}, "mem_gb": 15.83} +[eval step 500] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as \\( A \\):\n\n\\[ A = \\begin{bmatrix}\n1" +checkpoint snapshot queued -> outputs/healed/grid_general_fairness/reap_keep50_s1224_long500/step0500 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▃▃▅▄▆▁▃▅▅▄▂▂▂▄▄▁▃▄▅▂▃▇▂▂█▃█▂▃▁▃▄▃▃▄▃▂▄▅▃ +wandb: cumulative_loss_tokens ▁▁▁▂▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▄▄▅▅▅▅▅▅▆▆▇▇▇▇▇▇███ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅███████ +wandb: finish_rate ███▄▅▇██▇▃▇▇█▇█▇██▅▇██▆█▁█▅▇█▄▆██▇▆███▇▄ +wandb: forward_topk_kl █▃▄▂▃▅▂▅▄▆▄▃▃▄▅▅▄▂▄▂▁▃▃▃▂▄▂▄▂▁▂▃▃▂▂▁▂▃▃▃ +wandb: grad_norm █▂▂▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: mem_gb ▆▁▄▇█▆▂▇▆▃▇▃▄▆▅▆▆▆▄▃▂▇▃█▂▅▃▃▃▂█▇█▇▄█▇▂▆▅ +wandb: step ▁▁▂▂▂▂▂▂▂▂▂▃▃▃▃▄▄▄▄▄▅▅▅▅▆▆▆▆▆▇▇▇▇▇██████ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 313.3 +wandb: cumulative_loss_tokens 60000000 +wandb: epoch 2 +wandb: finish_rate 0.997 +wandb: forward_topk_kl 0.24799 +wandb: grad_norm 0.39258 +wandb: lr 3e-05 +wandb: mem_gb 15.83 +wandb: step 500 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run reap_keep50_s1224_long500 at: https://wandb.ai/hbfreed/glean-general-grid/runs/vehg5sig +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-general-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_general_fairness/reap_keep50_s1224_long500/wandb/run-20260719_142008-vehg5sig/logs diff --git a/healed/grid_general_fairness/reap_keep50_s1224_long500.eval.log b/healed/grid_general_fairness/reap_keep50_s1224_long500.eval.log new file mode 100644 index 0000000000000000000000000000000000000000..ff7ff6b90482242feddacb64dd1ebf076224c016 --- /dev/null +++ b/healed/grid_general_fairness/reap_keep50_s1224_long500.eval.log @@ -0,0 +1,140 @@ +2026-07-19T22:48:52-07:00 serving outputs/healed/grid_general_fairness/reap_keep50_s1224_long500/step0150 on GPU 1 port 8421 +2026-07-19T22:48:52-07:00 waiting for server /health ... +2026-07-19T22:49:22-07:00 server up; chat pass [gsm8k_cot_zeroshot,minerva_math500,ifeval] +2026-07-19:22:49:29 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot', 'minerva_math500', 'ifeval'] +2026-07-19:22:49:30 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234 +2026-07-19:22:49:30 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding! +2026-07-19:22:49:30 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8421/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3} +2026-07-19:22:49:30 INFO [models.api_models:179] Using max length 2048 - 1 +2026-07-19:22:49:30 INFO [models.api_models:200] Using tokenizer None +2026-07-19:22:49:35 INFO [evaluator_utils:446] Selected tasks: +2026-07-19:22:49:35 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml) +2026-07-19:22:49:35 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml) +2026-07-19:22:49:35 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml) +2026-07-19:22:49:35 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280} +2026-07-19:22:49:35 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-19:22:49:35 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-19:22:49:35 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0... + 0%| | 0/1319 [00:00 outputs/evals/general_suite/healed/reap_keep50_s1224_long500_step150 +2026-07-19T22:58:20-07:00 serving outputs/healed/grid_general_fairness/reap_keep50_s1224_long500/step0500 on GPU 1 port 8421 +2026-07-19T22:58:20-07:00 waiting for server /health ... +2026-07-19T22:58:45-07:00 server up; chat pass [gsm8k_cot_zeroshot,minerva_math500,ifeval] +2026-07-19:22:58:52 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot', 'minerva_math500', 'ifeval'] +2026-07-19:22:58:54 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234 +2026-07-19:22:58:54 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding! +2026-07-19:22:58:54 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8421/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3} +2026-07-19:22:58:54 INFO [models.api_models:179] Using max length 2048 - 1 +2026-07-19:22:58:54 INFO [models.api_models:200] Using tokenizer None +2026-07-19:22:58:59 INFO [evaluator_utils:446] Selected tasks: +2026-07-19:22:58:59 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml) +2026-07-19:22:58:59 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml) +2026-07-19:22:58:59 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml) +2026-07-19:22:58:59 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280} +2026-07-19:22:58:59 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-19:22:58:59 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-19:22:58:59 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0... + 0%| | 0/1319 [00:00 outputs/evals/general_suite/healed/reap_keep50_s1224_long500_step500 diff --git a/healed/grid_general_fairness/reap_keep50_s1224_lr1e5.console.log b/healed/grid_general_fairness/reap_keep50_s1224_lr1e5.console.log new file mode 100644 index 0000000000000000000000000000000000000000..9ba6a96f16f4fdfe4efa8892ca7ca72def902b7b --- /dev/null +++ b/healed/grid_general_fairness/reap_keep50_s1224_lr1e5.console.log @@ -0,0 +1,213 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_general_fairness/reap_keep50_s1224_lr1e5/wandb/run-20260719_141738-sgv88kwj +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run reap_keep50_s1224_lr1e5 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-general-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-general-grid/runs/sgv88kwj + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/grid_general_fairness/reap_keep50_s1224_lr1e5/step0150 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▄▃▁▄▃▂▂▂▁▇█▃▄▄▃▄▆▄▂▅▃▄▅▃▃▃▄▆▅▂▂▃▂▃▅▃▆▇▅▂ +wandb: cumulative_loss_tokens ▁▁▁▂▂▂▂▂▂▂▃▃▃▄▄▄▄▄▄▄▅▅▅▅▆▆▆▆▆▆▆▇▇▇▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate ██▄▄▇▃█▄▃▇█▇▆█▇█▆▅█▇▇▇▇▃█▇█▁██▆██▆▇▇██▆█ +wandb: forward_topk_kl ▇▇█▄▆▃▆▆▆▁▂▃▄▄▄▅▃▄▂▂▂▃▂▄▅▃▃▃▄▃▂▄▄▄▂▂▄▄▃▄ +wandb: grad_norm █▅▃▂▃▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▃██████████████████████████████████████ +wandb: mem_gb ▆▇▅▇▆▃▇▅▇▅▆▆▃▅▅▇▃▇▃▇▆▄▃▄▆▆▃▅▇▂▇▁█▂▇▂▁▇▂▆ +wandb: step ▁▁▂▂▂▂▃▃▃▃▃▃▃▄▄▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇████ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 405.4 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 0 +wandb: finish_rate 0.97 +wandb: forward_topk_kl 0.35079 +wandb: grad_norm 0.57031 +wandb: lr 1e-05 +wandb: mem_gb 15.93 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run reap_keep50_s1224_lr1e5 at: https://wandb.ai/hbfreed/glean-general-grid/runs/sgv88kwj +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-general-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_general_fairness/reap_keep50_s1224_lr1e5/wandb/run-20260719_141738-sgv88kwj/logs diff --git a/healed/grid_general_fairness/reap_keep50_s1224_lr1e5.eval.log b/healed/grid_general_fairness/reap_keep50_s1224_lr1e5.eval.log new file mode 100644 index 0000000000000000000000000000000000000000..8f0caa2d96a5ee4313388ff0678087d91503f8e4 --- /dev/null +++ b/healed/grid_general_fairness/reap_keep50_s1224_lr1e5.eval.log @@ -0,0 +1,70 @@ +2026-07-19T17:17:10-07:00 serving outputs/healed/grid_general_fairness/reap_keep50_s1224_lr1e5/step0150 on GPU 0 port 8420 +2026-07-19T17:17:10-07:00 waiting for server /health ... +2026-07-19T17:17:35-07:00 server up; chat pass [gsm8k_cot_zeroshot,minerva_math500,ifeval] +2026-07-19:17:17:42 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot', 'minerva_math500', 'ifeval'] +2026-07-19:17:17:43 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234 +2026-07-19:17:17:43 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding! +2026-07-19:17:17:43 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8420/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3} +2026-07-19:17:17:43 INFO [models.api_models:179] Using max length 2048 - 1 +2026-07-19:17:17:43 INFO [models.api_models:200] Using tokenizer None +2026-07-19:17:17:49 INFO [evaluator_utils:446] Selected tasks: +2026-07-19:17:17:49 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml) +2026-07-19:17:17:49 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml) +2026-07-19:17:17:49 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml) +2026-07-19:17:17:49 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280} +2026-07-19:17:17:49 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-19:17:17:49 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280} +2026-07-19:17:17:49 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0... + 0%| | 0/1319 [00:00 outputs/evals/general_suite/healed/reap_keep50_s1224_lr1e5_step150 diff --git a/healed/grid_math/glean_keep25_s1225.console.log b/healed/grid_math/glean_keep25_s1225.console.log new file mode 100644 index 0000000000000000000000000000000000000000..2a95945778f25660c7473efc28a5d7117e6d4ffc --- /dev/null +++ b/healed/grid_math/glean_keep25_s1225.console.log @@ -0,0 +1,230 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/glean_keep25_s1225/wandb/run-20260716_034843-v3mah2z8 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run glean-math-keep25-s1225 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/v3mah2z8 +12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 2.09B | teacher overlap=False +{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.3958471946775912, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 105.0, "lr": 6e-06, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 191}, "mem_gb": 9.94} +The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results. +[eval step 1] sample: '\nThe answer is **29**\n\nThe numbers are **1**, **3**, **5**, **7**. The four numbers are **1**, **3**, **5**, **7**. The total number of these four numbers is **15**. The number **29** is prime.\n\n**29*' +{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2812339077095192, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 92.5, "lr": 9e-06, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.22591313500156, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 62.75, "lr": 1.2e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 207}, "mem_gb": 10.0} +{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.9309642657568057, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 14.9375, "lr": 1.5e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 208}, "mem_gb": 9.96} +{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7706553599588573, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 10.9375, "lr": 1.8e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6402923999081055, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 10.9375, "lr": 2.1e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 246}, "mem_gb": 9.87} +{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5687407882797222, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 3.71875, "lr": 2.4e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.6, "frames": {"chat": 206}, "mem_gb": 10.02} +{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.47117147324259084, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 2.46875, "lr": 2.7000000000000002e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 233}, "mem_gb": 10.0} +{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.43779878526590765, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 1.875, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 229}, "mem_gb": 9.87} +{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3573542028216024, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 1.3515625, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.2, "frames": {"chat": 236}, "mem_gb": 9.9} +[eval step 10] sample: "To solve this problem, we need to determine which of the four numbers on the diagonal from \\(7\\) to \\(49\\) are prime. Let's break down the steps:\n\n1. **Identify the numbers on the diagonal:**\n The n" +{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3389050622591128, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 1.140625, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 239}, "mem_gb": 9.79} +{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2907669429977735, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 0.95703125, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 241}, "mem_gb": 9.91} +{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2989568690617879, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.98828125, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 226}, "mem_gb": 9.87} +{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25473856463196376, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.77734375, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.4, "frames": {"chat": 234}, "mem_gb": 10.0} +{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2544649389738838, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.74609375, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 257}, "mem_gb": 9.99} +{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3264326198320836, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.953125, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 208}, "mem_gb": 10.05} +{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2732485909540206, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.8515625, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.2, "frames": {"chat": 211}, "mem_gb": 10.02} +{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24720005844794213, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.765625, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 227}, "mem_gb": 10.0} +{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25496282017032307, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.7734375, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 211}, "mem_gb": 9.98} +{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21520358278726537, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.65234375, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 238}, "mem_gb": 10.0} +[eval step 20] sample: 'To solve this problem, we need to identify the four numbers that lie on the diagonal from the center of the grid to the number \\(7\\) in the given sequence. The sequence is a spiral pattern starting at' +{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2216773895036429, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.5, "frames": {"chat": 237}, "mem_gb": 10.04} +{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24041704051339377, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.68359375, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 208}, "mem_gb": 10.04} +{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2146310205236698, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.59375, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 221}, "mem_gb": 10.12} +{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20238777070970584, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.5859375, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.6, "frames": {"chat": 232}, "mem_gb": 9.96} +{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20690553727087876, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 208}, "mem_gb": 9.99} +{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18880456114976357, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.0, "frames": {"chat": 227}, "mem_gb": 9.91} +{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20486585594148685, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.5625, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 221}, "mem_gb": 9.94} +{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18253140152214717, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.2, "frames": {"chat": 232}, "mem_gb": 10.01} +{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17699584313053637, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 219}, "mem_gb": 10.01} +{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19838463523512084, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 195}, "mem_gb": 10.1} +[eval step 30] sample: 'To solve this problem, we need to analyze the arrangement of numbers on a square grid starting from the center and then determine how many of the numbers on the diagonal from \\(7\\) are prime.\n\n### Ste' +{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18508145255607863, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 216}, "mem_gb": 10.0} +{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15833309386118005, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 208}, "mem_gb": 9.89} +{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15909730202872305, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 235}, "mem_gb": 9.88} +{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.162696361270609, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 225}, "mem_gb": 9.99} +{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22678135332918414, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 213}, "mem_gb": 10.08} +{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15918074906071028, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.2, "frames": {"chat": 257}, "mem_gb": 9.76} +{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18729867079984397, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 212}, "mem_gb": 10.03} +{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15989416230500986, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 221}, "mem_gb": 10.0} +{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16033389104095597, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.4, "frames": {"chat": 242}, "mem_gb": 10.0} +{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.14507702404971545, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 220}, "mem_gb": 9.96} +[eval step 40] sample: 'To solve this problem, we need to identify the four numbers from the set \\(\\{1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23,' +{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15030326494518667, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 240}, "mem_gb": 9.86} +{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1607294344684109, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 206}, "mem_gb": 9.99} +{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1607213422082675, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 226}, "mem_gb": 10.0} +{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.201115768094485, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.8046875, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.4, "frames": {"chat": 244}, "mem_gb": 9.79} +{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17653826248689244, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.4, "frames": {"chat": 224}, "mem_gb": 10.01} +{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1541471158782641, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 271}, "mem_gb": 9.73} +{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.14582899590842427, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 236}, "mem_gb": 10.01} +{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1692912352339054, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.90625, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 232}, "mem_gb": 9.88} +{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13167550868373365, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 210}, "mem_gb": 9.94} +{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.12800594935423384, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 217}, "mem_gb": 9.9} +[eval step 50] sample: 'To solve this problem, we need to analyze the numbers arranged in a spiral pattern on a square grid and determine how many of the four numbers that lie on the same diagonal as the number \\(7\\) are pri' +checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep25_s1225/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16889489130682, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.490234375, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 224}, "mem_gb": 10.02} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19071834716852754, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.4, "frames": {"chat": 203}, "mem_gb": 9.87} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15086872272131344, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 239}, "mem_gb": 9.97} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11240008413158357, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 254}, "mem_gb": 9.88} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11910581262825677, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 241}, "mem_gb": 9.98} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15324119401192293, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 213}, "mem_gb": 10.01} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12598269802760334, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 221}, "mem_gb": 10.05} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1358413405182461, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 196}, "mem_gb": 10.01} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11328371758495147, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 270}, "mem_gb": 9.82} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10341172415204346, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 216}, "mem_gb": 9.99} +[eval step 60] sample: 'To solve this problem, we need to analyze the arrangement of numbers on a square grid and determine how many of the four numbers on the same diagonal as the number \\(7\\) are prime.\n\n### Steps to Solve' +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14843227218265334, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 200}, "mem_gb": 9.96} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09767647584409764, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 206}, "mem_gb": 9.91} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09538950520747652, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 234}, "mem_gb": 9.95} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11610429440625011, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 215}, "mem_gb": 9.96} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09513266892942289, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 255}, "mem_gb": 9.94} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11722616833361486, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 250}, "mem_gb": 9.82} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10852097176260625, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.2, "frames": {"chat": 242}, "mem_gb": 9.99} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14525317518096417, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16521108877193183, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 208}, "mem_gb": 10.04} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1191304967135191, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 208}, "mem_gb": 9.97} +[eval step 70] sample: 'To solve this problem, we need to analyze the spiral pattern on the square grid and determine the prime numbers among the numbers that appear in the shaded squares.\n\n### Steps to Solve:\n\n1. **Understa' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12873777522143598, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 209}, "mem_gb": 10.12} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09841357960260162, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 235}, "mem_gb": 9.96} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09810730254290005, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 204}, "mem_gb": 9.95} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.137890988603731, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10093486693870897, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.8, "frames": {"chat": 240}, "mem_gb": 10.0} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11269029565745343, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 246}, "mem_gb": 9.99} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11716039955631519, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.439453125, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 243}, "mem_gb": 9.81} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14083528584881375, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11617199123160293, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13478537587194392, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 211}, "mem_gb": 10.01} +[eval step 80] sample: 'To solve this problem, we need to analyze the arrangement of numbers on a square grid and determine how many of the numbers on the same diagonal as the number \\(7\\) are prime.\n\n### Steps to Solve the ' +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10429689287121097, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 232}, "mem_gb": 9.97} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12193509586900472, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 214}, "mem_gb": 10.01} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10920041576723258, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 226}, "mem_gb": 9.9} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1078172554230628, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09039803395358224, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 1.125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 218}, "mem_gb": 9.83} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09356577665278067, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12304461509076257, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.4, "frames": {"chat": 215}, "mem_gb": 10.01} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1115369677213952, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09695998856667429, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 209}, "mem_gb": 9.94} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09343226164858788, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.1, "frames": {"chat": 262}, "mem_gb": 9.87} +[eval step 90] sample: 'To solve this problem, we need to analyze the arrangement of numbers on a square grid and determine how many of the numbers in the shaded squares on the same diagonal as the number \\(7\\) are prime.\n\nL' +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0932332858821998, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.3, "frames": {"chat": 249}, "mem_gb": 9.96} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1266877316886559, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.7, "frames": {"chat": 227}, "mem_gb": 10.0} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10181538004527489, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 221}, "mem_gb": 9.99} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10009486745695273, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 234}, "mem_gb": 10.01} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08687792688549185, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 213}, "mem_gb": 9.96} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08524717839102572, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 213}, "mem_gb": 9.89} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10703749001209313, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 234}, "mem_gb": 9.92} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10287342213264977, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.0, "frames": {"chat": 222}, "mem_gb": 9.99} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11010189882616202, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 227}, "mem_gb": 10.01} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10967375374240801, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 218}, "mem_gb": 10.04} +[eval step 100] sample: "To solve this problem, we need to understand the arrangement of numbers on the square grid and how the numbers are placed based on their positions. Here's a step-by-step approach:\n\n1. **Understand the" +checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep25_s1225/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11762644656381259, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 223}, "mem_gb": 10.01} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11568864874821157, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 206}, "mem_gb": 10.01} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09638762137287607, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 213}, "mem_gb": 9.92} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13616388885583727, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 223}, "mem_gb": 9.86} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1067974968187511, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 227}, "mem_gb": 9.97} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09873509239495422, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 253}, "mem_gb": 10.0} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11062951422066739, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 216}, "mem_gb": 10.01} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08701037775237734, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 205}, "mem_gb": 9.98} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11230816109282896, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 207}, "mem_gb": 10.06} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09889225037389746, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 215}, "mem_gb": 9.98} +[eval step 110] sample: "To solve this problem, we need to analyze the spiral pattern of numbers on a square grid and determine which of the four shaded squares contain prime numbers. Here's a step-by-step approach:\n\n1. **Und" +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07409078115209317, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.0, "frames": {"chat": 228}, "mem_gb": 10.0} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08839880903875455, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.4, "frames": {"chat": 221}, "mem_gb": 10.04} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06728584335437045, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 254}, "mem_gb": 9.84} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06614433599614228, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 210}, "mem_gb": 9.97} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07396190701468537, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 226}, "mem_gb": 9.92} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08188189537149544, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 212}, "mem_gb": 9.99} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09220475707668811, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 211}, "mem_gb": 9.93} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0956091513237295, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 196}, "mem_gb": 9.97} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07131106523716202, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 212}, "mem_gb": 10.0} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0689470173562877, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 244}, "mem_gb": 9.91} +[eval step 120] sample: "To solve this problem, we need to analyze the spiral pattern of numbers on a square grid and determine which of the four shaded squares contain prime numbers. Here's a step-by-step approach:\n\n1. **Und" +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07394362696452687, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 1.6484375, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 222}, "mem_gb": 9.95} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07991834989693015, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 218}, "mem_gb": 10.0} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08040042725556219, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.3, "frames": {"chat": 252}, "mem_gb": 9.88} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08868840474116926, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 202}, "mem_gb": 10.05} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09232083391323685, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 237}, "mem_gb": 10.0} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08302971795774065, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.298828125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.7, "frames": {"chat": 234}, "mem_gb": 9.99} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0661788280802158, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 215}, "mem_gb": 10.0} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0645129962954515, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.0, "frames": {"chat": 234}, "mem_gb": 9.93} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06428662312957459, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 216}, "mem_gb": 9.99} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07294256055513397, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 210}, "mem_gb": 9.95} +[eval step 130] sample: "To solve this problem, we need to analyze the spiral pattern of numbers on a square grid and determine which of the four shaded squares contain prime numbers. Here's a step-by-step approach:\n\n1. **Und" +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08209835358545339, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07414536922099069, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07274689665936554, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 225}, "mem_gb": 9.95} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07239032591326783, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.3, "frames": {"chat": 253}, "mem_gb": 9.85} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0755888467406854, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 247}, "mem_gb": 9.97} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0824168190576757, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 238}, "mem_gb": 9.97} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09111157214685033, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 235}, "mem_gb": 10.0} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08819460893472035, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 215}, "mem_gb": 9.97} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07084991472096493, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.298828125, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.4, "frames": {"chat": 268}, "mem_gb": 9.97} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07839255714469279, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.4, "frames": {"chat": 228}, "mem_gb": 10.0} +[eval step 140] sample: "To solve this problem, we need to analyze the spiral pattern of numbers on the square grid and determine which of the four shaded squares contain prime numbers. Here's a step-by-step approach:\n\n1. **U" +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07923957366729155, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 252}, "mem_gb": 9.93} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07850046219552556, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 223}, "mem_gb": 10.01} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0933897839378876, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.6, "frames": {"chat": 226}, "mem_gb": 10.0} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09283627185517301, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.1, "frames": {"chat": 208}, "mem_gb": 10.05} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06709443831786824, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.7, "frames": {"chat": 240}, "mem_gb": 9.93} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08292833760709813, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 222}, "mem_gb": 9.93} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06971678353363338, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 4.125, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 236}, "mem_gb": 10.0} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06898172454525096, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 217}, "mem_gb": 9.96} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07392021829135095, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 252}, "mem_gb": 9.88} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0710504538111544, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 222}, "mem_gb": 9.99} +[eval step 150] sample: "To solve this problem, we need to analyze the spiral pattern of numbers on a square grid and determine which of the four shaded squares contain prime numbers. Here's a step-by-step approach:\n\n1. **Und" +checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep25_s1225/step0150 +wandb: updating run metadata +wandb: uploading summary, console lines 170-170 +wandb: +wandb: Run history: +wandb: comp_len █▆▂▃▃▅▃▆▄▃▇▁▆▄▃▄▆▃▄▆▆▆▂▅▄▃▅▅▄▄▅▂▄▁▅▆▄▃▁▅ +wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅█████████████ +wandb: finish_rate █▆▆▃▆▄▅▁▃█▆▅▇▂▃▅▂▂▅▇▄▃▆▄▅▇▄▇▄▅▄▃▂▆█▆█▄█▅ +wandb: forward_topk_kl █▃▃▂▂▃▂▂▂▂▂▂▂▂▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm █▇▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▅█████████████████████████████████████ +wandb: mem_gb ▆▄▇▄▇█▁▇▃▆▄▅▄▆▇▂▇▅▆▇▃▆▇▆▅▇▅▃▇██▅▄█▇▇▆▇▅▄ +wandb: step ▁▁▁▁▁▂▂▂▂▂▃▃▄▄▄▄▄▄▅▅▅▅▅▅▅▆▆▆▆▆▆▆▇▇▇▇████ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 540.5 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.847 +wandb: forward_topk_kl 0.07105 +wandb: grad_norm 0.28516 +wandb: lr 3e-05 +wandb: mem_gb 9.99 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run glean-math-keep25-s1225 at: https://wandb.ai/hbfreed/glean-grid/runs/v3mah2z8 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/glean_keep25_s1225/wandb/run-20260716_034843-v3mah2z8/logs +{ + "correct": 560, + "accuracy": 0.4245640636846095, + "finished": 1259, + "finish_rate": 0.954510993176649, + "mean_completion_tokens": 173.51023502653524 +} +saved item-level results -> outputs/evals/grid_math/glean_keep25_s1225_step100_chat.json +{ + "correct": 564, + "accuracy": 0.4275966641394996, + "finished": 1270, + "finish_rate": 0.9628506444275967, + "mean_completion_tokens": 173.10007581501137 +} +saved item-level results -> outputs/evals/grid_math/glean_keep25_s1225_step150_chat.json diff --git a/healed/grid_math/glean_keep25_s1226.console.log b/healed/grid_math/glean_keep25_s1226.console.log new file mode 100644 index 0000000000000000000000000000000000000000..321c12d9a2196795d06bddb9ca12e7cc63836ab1 --- /dev/null +++ b/healed/grid_math/glean_keep25_s1226.console.log @@ -0,0 +1,232 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run vhgtf0ej +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/glean_keep25_s1226/wandb/run-20260716_034735-vhgtf0ej +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run glean-math-keep25-s1226 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/vhgtf0ej +12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 2.09B | teacher overlap=False +{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.4740517740617196, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 121.0, "lr": 6e-06, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.0, "frames": {"chat": 254}, "mem_gb": 9.82} +The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results. +[eval step 1] sample: 'The value is 28, and the perimeter of the resulting triangle is 28. The problem is to find the perimeter of the resulting triangle, given the conditions mentioned.\n\nThe perimeter of the resulting tria' +{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.377854461752375, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 95.0, "lr": 9e-06, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 241}, "mem_gb": 9.98} +{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2003990252320964, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 59.75, "lr": 1.2e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 213}, "mem_gb": 10.01} +{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8684858515068888, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 14.8125, "lr": 1.5e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.0, "frames": {"chat": 221}, "mem_gb": 10.05} +{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.841784818589439, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 8.3125, "lr": 1.8e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 196}, "mem_gb": 10.01} +{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6113010039225221, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 8.375, "lr": 2.1e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 270}, "mem_gb": 9.82} +{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5845785550904771, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 4.53125, "lr": 2.4e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 216}, "mem_gb": 9.99} +{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.49846355576987067, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 3.390625, "lr": 2.7000000000000002e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 200}, "mem_gb": 9.96} +{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4226150450040897, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 2.609375, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 206}, "mem_gb": 9.91} +{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3381338079671065, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 1.7421875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 234}, "mem_gb": 9.95} +[eval step 10] sample: "To solve this problem, we need to determine the lengths of the sides of the triangle given the perimeter and the midpoints of its sides. Let's break down the problem into manageable steps:\n\n1. **Under" +{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.34357373327190677, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 1.3828125, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 215}, "mem_gb": 9.96} +{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2818106052008768, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 1.1484375, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 255}, "mem_gb": 9.94} +{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27427206053423386, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.91796875, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.4, "frames": {"chat": 250}, "mem_gb": 9.82} +{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2693552000547449, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.8359375, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 242}, "mem_gb": 9.99} +{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.29865186369282504, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.8671875, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3619642677543064, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.99609375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 208}, "mem_gb": 10.04} +{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25310858338586983, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.81640625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 208}, "mem_gb": 9.97} +{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2819262358199805, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.80078125, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 209}, "mem_gb": 10.12} +{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2152011162099739, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.71875, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 235}, "mem_gb": 9.96} +{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21725329664709667, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.7265625, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.6, "frames": {"chat": 204}, "mem_gb": 9.95} +[eval step 20] sample: 'To solve this problem, we need to understand the properties of a triangle and its midpoints.\n\n1. **Understanding the Properties:**\n - The perimeter of a triangle is given by the sum of its sides.\n ' +{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2732300681939969, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.87890625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20615510911339274, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 240}, "mem_gb": 10.0} +{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19796610735058784, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 246}, "mem_gb": 9.99} +{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21532276370519152, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.703125, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.4, "frames": {"chat": 243}, "mem_gb": 9.81} +{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2322601548684761, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.6875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.6, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21518369818814098, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23441068366014708, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.6953125, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 211}, "mem_gb": 10.01} +{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1949662358665218, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.2, "frames": {"chat": 232}, "mem_gb": 9.97} +{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21759224594825258, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.6171875, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 214}, "mem_gb": 10.01} +{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18679429283955445, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 226}, "mem_gb": 9.9} +[eval step 30] sample: 'To solve this problem, we need to understand the properties of a triangle and its midpoints.\n\n1. **Understand the Properties:**\n - The perimeter of a triangle is the sum of its side lengths.\n - Th' +{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17500035125799476, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16020464946124702, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 218}, "mem_gb": 9.83} +{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17072588143019626, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2061376795306181, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 215}, "mem_gb": 10.01} +{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18651440346712866, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16298052088283002, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 209}, "mem_gb": 9.94} +{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15828031261581926, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 262}, "mem_gb": 9.87} +{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1595262515474111, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 249}, "mem_gb": 9.96} +{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19758843879966687, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 227}, "mem_gb": 10.0} +{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.14780844743441168, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 221}, "mem_gb": 9.99} +[eval step 40] sample: "To solve this problem, we need to understand the properties of the midpoints of a triangle's sides and how they affect the perimeter.\n\n1. **Midpoints of Sides:**\n - The midpoint of a side of length " +{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16207660000277682, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 234}, "mem_gb": 10.01} +{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13810028650884826, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 213}, "mem_gb": 9.96} +{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13506372714247555, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 213}, "mem_gb": 9.89} +{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.14545865754342327, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 234}, "mem_gb": 9.92} +{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.14481995174276333, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 222}, "mem_gb": 9.99} +{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17587772664371878, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 227}, "mem_gb": 10.01} +{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1550076254642258, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 218}, "mem_gb": 10.04} +{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17269171449063966, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 223}, "mem_gb": 10.01} +{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17547114912066608, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 206}, "mem_gb": 10.01} +{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1537847354217743, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.490234375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 213}, "mem_gb": 9.92} +[eval step 50] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is given as" +checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep25_s1226/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18943238889773686, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 223}, "mem_gb": 9.86} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16693324066617837, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 227}, "mem_gb": 9.97} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.14277432735928644, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.6, "frames": {"chat": 253}, "mem_gb": 10.0} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1437762385570444, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.83984375, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 216}, "mem_gb": 10.01} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1300227014787495, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 205}, "mem_gb": 9.98} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15664660246831674, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.8, "frames": {"chat": 207}, "mem_gb": 10.06} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1445078781637363, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 215}, "mem_gb": 9.98} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10567276090113446, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 228}, "mem_gb": 10.0} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13750520292868218, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.3, "frames": {"chat": 221}, "mem_gb": 10.04} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10064160097377996, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 254}, "mem_gb": 9.84} +[eval step 60] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is given as" +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09463851169059052, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 210}, "mem_gb": 9.97} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10595882567154864, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.2, "frames": {"chat": 226}, "mem_gb": 9.92} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12365473369524503, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 212}, "mem_gb": 9.99} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12891767050419004, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.7, "frames": {"chat": 211}, "mem_gb": 9.93} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13171432805840547, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 196}, "mem_gb": 9.97} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11064968370975306, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 212}, "mem_gb": 10.0} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1052026781039002, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 244}, "mem_gb": 9.91} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10334055133834481, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 222}, "mem_gb": 9.95} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11475445196203267, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 218}, "mem_gb": 10.0} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12215668223105992, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.2, "frames": {"chat": 252}, "mem_gb": 9.88} +[eval step 70] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by the midpoints of the sides of the original triangle.\n\n1. **Identify the Original Triangle:**\n Let the ' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13357413773555307, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 202}, "mem_gb": 10.05} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15701979057999949, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.4, "frames": {"chat": 237}, "mem_gb": 10.0} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11951561769278099, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 234}, "mem_gb": 9.99} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09150583964148536, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 215}, "mem_gb": 10.0} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08957106544353689, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 234}, "mem_gb": 9.93} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08833280240970974, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 216}, "mem_gb": 9.99} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1060180736657232, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 210}, "mem_gb": 9.95} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10358950277809054, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09346498899217695, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10706426500252758, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 225}, "mem_gb": 9.95} +[eval step 80] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and its midpoints.\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is given as 28.\n ' +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10597167926300317, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 253}, "mem_gb": 9.85} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10689509418696785, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.6, "frames": {"chat": 247}, "mem_gb": 9.97} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10978524073281636, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.4, "frames": {"chat": 238}, "mem_gb": 9.97} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11432554268656919, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 235}, "mem_gb": 10.0} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11262670917728294, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 215}, "mem_gb": 9.97} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11482771853323405, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 268}, "mem_gb": 9.97} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10706278533112878, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 228}, "mem_gb": 10.0} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11632729308865965, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 252}, "mem_gb": 9.93} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09904835979019602, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 223}, "mem_gb": 10.01} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15139277683890734, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.6, "frames": {"chat": 226}, "mem_gb": 10.0} +[eval step 90] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by the midpoints of the sides of a triangle with a given perimeter.\n\n### Steps to Solve:\n\n1. **Understand t' +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12696473920823384, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.0, "frames": {"chat": 208}, "mem_gb": 10.05} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0981225883629794, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.6, "frames": {"chat": 240}, "mem_gb": 9.93} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11373697855863721, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 222}, "mem_gb": 9.93} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09599229777337362, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 236}, "mem_gb": 10.0} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0905342026155442, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 217}, "mem_gb": 9.96} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10233824915119136, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 252}, "mem_gb": 9.88} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09226011450669418, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 222}, "mem_gb": 9.99} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1017231995769466, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.901, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.4, "frames": {"chat": 242}, "mem_gb": 9.87} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13698266774720202, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 219}, "mem_gb": 9.93} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09786459891942019, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 223}, "mem_gb": 9.94} +[eval step 100] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.\n -" +checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep25_s1226/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11257973559337357, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 232}, "mem_gb": 9.95} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11817761343022187, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.832, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 220}, "mem_gb": 10.0} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1162180175398166, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 210}, "mem_gb": 10.04} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10464485003838005, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 226}, "mem_gb": 9.97} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09739518145142743, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.741, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 212}, "mem_gb": 10.0} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09368765245955438, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 236}, "mem_gb": 10.01} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07762365770225103, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.928, "comp_len": 454.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 264}, "mem_gb": 9.88} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1221225647800602, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.52734375, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 229}, "mem_gb": 9.98} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0774544737749733, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 465.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.3, "frames": {"chat": 258}, "mem_gb": 9.85} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11255108813242987, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 208}, "mem_gb": 10.01} +[eval step 110] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by the midpoints of the sides of the original triangle.\n\n1. **Midpoints of a Triangle:**\n - The midpoint ' +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0868233092402108, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.88, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.7, "frames": {"chat": 249}, "mem_gb": 9.93} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07358991829988858, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 220}, "mem_gb": 9.99} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08318625353276729, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 223}, "mem_gb": 9.99} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07259801399639497, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 221}, "mem_gb": 10.0} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07171069395930196, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 230}, "mem_gb": 9.9} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08987705074110999, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 214}, "mem_gb": 9.97} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11477152344050506, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 214}, "mem_gb": 9.99} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08955526669428994, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 210}, "mem_gb": 10.04} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09731406231168657, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.6, "frames": {"chat": 214}, "mem_gb": 10.0} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0877075838214097, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.791, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 215}, "mem_gb": 9.95} +[eval step 120] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the triangle is 28.\n - The midp" +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09859609124142056, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 208}, "mem_gb": 9.99} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07579588066336389, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.789, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 218}, "mem_gb": 9.88} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06975389175747211, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 233}, "mem_gb": 9.9} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06521768421179926, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 231}, "mem_gb": 9.94} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08123537786286324, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 235}, "mem_gb": 10.13} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07916606919132173, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 216}, "mem_gb": 9.99} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0735622343561612, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.831, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.4, "frames": {"chat": 237}, "mem_gb": 10.01} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10053933701403439, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.4, "frames": {"chat": 203}, "mem_gb": 10.01} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07754695314977629, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.873, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 236}, "mem_gb": 10.04} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07985588553401952, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 214}, "mem_gb": 10.01} +[eval step 130] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by the midpoints of the sides of a given triangle.\n\n1. **Understand the Midpoints:**\n - The midpoint of a' +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0826984562702477, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 209}, "mem_gb": 10.0} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07694104558015243, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 256}, "mem_gb": 10.0} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07371389015351111, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 230}, "mem_gb": 9.87} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07886174646293123, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.6, "frames": {"chat": 230}, "mem_gb": 10.06} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09088345281931882, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 227}, "mem_gb": 9.95} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09291109785580387, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09028420611228793, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.699, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.2, "frames": {"chat": 206}, "mem_gb": 10.03} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08365592800673718, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.82, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 228}, "mem_gb": 9.9} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08592219653519181, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 224}, "mem_gb": 10.0} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07787073799719413, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.66, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 200}, "mem_gb": 10.03} +[eval step 140] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by the midpoints of the sides of the original triangle.\n\n1. **Understand the Midpoints:**\n The midpoints ' +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07876405616238868, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.714, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 196}, "mem_gb": 10.01} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0733729308873415, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 223}, "mem_gb": 9.99} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07316665141967435, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.869, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 213}, "mem_gb": 9.89} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06799636307867865, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 232}, "mem_gb": 9.93} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06842424086419245, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 223}, "mem_gb": 9.93} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07442820003066833, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.85, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 233}, "mem_gb": 10.02} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08168693628491212, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.816, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 217}, "mem_gb": 10.01} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11692778237918391, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.752, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 202}, "mem_gb": 10.08} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0718278338014769, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 253}, "mem_gb": 9.93} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06792547624123593, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 231}, "mem_gb": 9.94} +[eval step 150] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by the midpoints of the sides of the original triangle.\n\n1. **Midpoints of a Triangle:**\n The midpoint of' +checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep25_s1226/step0150 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: uploading summary, console lines 170-170 +wandb: +wandb: Run history: +wandb: comp_len █▆▂█▃▂▆▅▅▆▁▅▅▅▄▆▄▆▃▄▆▂▃▄▂▄▅▂▃▅▆▃▁▆▅▆▃▅▃▆ +wandb: cumulative_loss_tokens ▁▁▂▂▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇███ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅███████████████ +wandb: finish_rate ▇▇▃▆█▃▆▃▅▅▅▄▆▆▄▃▃▃▆▆▇▅▃▆▇▇▇▅▃█▇▆▄▆▆▇▂▅▁▆ +wandb: forward_topk_kl ██▆▅▄▂▂▂▂▂▂▂▂▂▂▂▂▂▂▁▁▁▁▁▁▁▁▁▂▁▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm █▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁███████████████████████████████████████ +wandb: mem_gb ▆▅▂▃▄▃▅▅▄▄▄▅▅▆▃▄▅▆▄▆▅▁▅▂▄▆▁▅▄▅▃█▄▅▅▅▅▂▅▃ +wandb: step ▁▁▁▁▁▂▂▂▂▂▂▂▃▃▃▃▃▃▃▄▄▄▄▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇██ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 519.5 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.879 +wandb: forward_topk_kl 0.06793 +wandb: grad_norm 0.26367 +wandb: lr 3e-05 +wandb: mem_gb 9.94 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run glean-math-keep25-s1226 at: https://wandb.ai/hbfreed/glean-grid/runs/vhgtf0ej +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/glean_keep25_s1226/wandb/run-20260716_034735-vhgtf0ej/logs +{ + "correct": 557, + "accuracy": 0.422289613343442, + "finished": 1276, + "finish_rate": 0.9673995451099318, + "mean_completion_tokens": 191.71645185746777 +} +saved item-level results -> outputs/evals/grid_math/glean_keep25_s1226_step100_chat.json +{ + "correct": 575, + "accuracy": 0.4359363153904473, + "finished": 1279, + "finish_rate": 0.9696739954510993, + "mean_completion_tokens": 193.40788476118271 +} +saved item-level results -> outputs/evals/grid_math/glean_keep25_s1226_step150_chat.json diff --git a/healed/grid_math/keep75.log b/healed/grid_math/keep75.log new file mode 100644 index 0000000000000000000000000000000000000000..1c2641a9b1c03fb9779e16e6517425b98c54843e --- /dev/null +++ b/healed/grid_math/keep75.log @@ -0,0 +1,56 @@ +2026-07-16T14:28:11-07:00 === keep-75 finish, 2 GPUs concurrent (3rd idle for power) === +2026-07-16T14:28:11-07:00 RESUMING glean_keep75_s1226 from step0100 (optimizer+scheduler+data position restored) +2026-07-16T14:28:11-07:00 RESUMING glean_keep75_s1225 from step0100 (optimizer+scheduler+data position restored) +2026-07-16T15:09:12-07:00 eval glean_keep75_s1225 step100 +2026-07-16T15:10:41-07:00 eval glean_keep75_s1226 step100 +2026-07-16T15:11:18-07:00 eval glean_keep75_s1225 step150 +2026-07-16T15:12:37-07:00 eval glean_keep75_s1226 step150 +2026-07-16T15:13:13-07:00 glean_keep75_s1225 done -> 0.689158453373768 +2026-07-16T15:13:13-07:00 healing uniform_keep75_s1224 on GPU-a6acf07f (port 8392) +2026-07-16T15:14:35-07:00 glean_keep75_s1226 done -> 0.7012888551933283 +2026-07-16T15:14:35-07:00 healing glean_keep75_s1224 on GPU-8ca70870 (port 8391) +2026-07-16T17:14:44-07:00 eval glean_keep75_s1224 step100 +2026-07-16T17:16:42-07:00 eval glean_keep75_s1224 step150 +2026-07-16T17:18:39-07:00 glean_keep75_s1224 done -> 0.6914329037149356 +2026-07-16T17:18:39-07:00 healing reap_keep75_s1224 on GPU-8ca70870 (port 8391) +2026-07-16T17:24:11-07:00 eval uniform_keep75_s1224 step100 +2026-07-16T17:26:08-07:00 eval uniform_keep75_s1224 step150 +2026-07-16T17:28:03-07:00 uniform_keep75_s1224 done -> 0.6338134950720242 +2026-07-16T17:28:03-07:00 healing reap_keep75_s1225 on GPU-a6acf07f (port 8392) +2026-07-16T19:38:42-07:00 eval reap_keep75_s1225 step100 +2026-07-16T19:40:26-07:00 eval reap_keep75_s1225 step150 +2026-07-16T19:42:01-07:00 reap_keep75_s1225 done -> 0.6755117513267627 +2026-07-16T19:42:01-07:00 healing uniform_keep75_s1226 on GPU-a6acf07f (port 8392) +2026-07-16T19:46:16-07:00 === keep-75 finish, 2 GPUs concurrent (3rd idle for power) === +2026-07-16T19:46:16-07:00 glean_keep75_s1225 already done, skip +2026-07-16T19:46:16-07:00 glean_keep75_s1226 already done, skip +2026-07-16T19:46:16-07:00 uniform_keep75_s1224 already done, skip +2026-07-16T19:46:16-07:00 glean_keep75_s1224 already done, skip +2026-07-16T19:46:16-07:00 reap_keep75_s1225 already done, skip +2026-07-16T19:46:16-07:00 RESUMING reap_keep75_s1224 from step0050 (optimizer+scheduler+data position restored) +2026-07-16T19:46:16-07:00 healing reap_keep75_s1226 on GPU-a6acf07f (port 8392) +2026-07-16T21:00:24-07:00 === keep-75 finish, 2 GPUs concurrent (3rd idle for power) === +2026-07-16T21:00:24-07:00 glean_keep75_s1226 already done, skip +2026-07-16T21:00:24-07:00 glean_keep75_s1225 already done, skip +2026-07-16T21:00:24-07:00 glean_keep75_s1224 already done, skip +2026-07-16T21:00:24-07:00 uniform_keep75_s1224 already done, skip +2026-07-16T21:00:24-07:00 RESUMING reap_keep75_s1224 from step0050 (optimizer+scheduler+data position restored) +2026-07-16T21:00:24-07:00 reap_keep75_s1225 already done, skip +2026-07-16T21:00:24-07:00 RESUMING reap_keep75_s1226 from step0050 (optimizer+scheduler+data position restored) +2026-07-16T22:27:41-07:00 eval reap_keep75_s1224 step100 +2026-07-16T22:28:47-07:00 eval reap_keep75_s1226 step100 +2026-07-16T22:29:21-07:00 eval reap_keep75_s1224 step150 +2026-07-16T22:30:31-07:00 eval reap_keep75_s1226 step150 +2026-07-16T22:30:57-07:00 reap_keep75_s1224 done -> 0.6846095526914329 +2026-07-16T22:30:57-07:00 healing uniform_keep75_s1225 on GPU-864c54df (port 8391) +2026-07-16T22:32:08-07:00 reap_keep75_s1226 done -> 0.6732373009855952 +2026-07-16T22:32:08-07:00 healing uniform_keep75_s1226 on GPU-a6acf07f (port 8392) +2026-07-17T00:28:54-07:00 eval uniform_keep75_s1226 step100 +2026-07-17T00:29:31-07:00 eval uniform_keep75_s1225 step100 +2026-07-17T00:30:52-07:00 eval uniform_keep75_s1226 step150 +2026-07-17T00:31:28-07:00 eval uniform_keep75_s1225 step150 +2026-07-17T00:32:51-07:00 uniform_keep75_s1226 done -> 0.6322971948445792 +2026-07-17T00:32:51-07:00 LANE GPU-a6acf07f complete +2026-07-17T00:33:23-07:00 uniform_keep75_s1225 done -> 0.640636846095527 +2026-07-17T00:33:23-07:00 LANE GPU-864c54df complete +2026-07-17T00:33:23-07:00 === KEEP75 COMPLETE === diff --git a/healed/grid_math/reap_keep25_s1224.console.log b/healed/grid_math/reap_keep25_s1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..3c4d5d393d5786bec91178ba0534bff468e5b3e7 --- /dev/null +++ b/healed/grid_math/reap_keep25_s1224.console.log @@ -0,0 +1,232 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run q0iymgs1 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/reap_keep25_s1224/wandb/run-20260716_071102-q0iymgs1 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run reap-math-keep25-s1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/q0iymgs1 +12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 2.09B | teacher overlap=False +{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 5.414990429035822, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 92.0, "lr": 6e-06, "finish_rate": 0.907, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 236}, "mem_gb": 9.77} +The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results. +[eval step 1] sample: '\nisk= \n\\\na <\n\n1. \n.ile = \n formula, eq.subsistry-nals,\n equr. ome,m. -Te \\\'\n dolect{for["cey"$-\ndiscabJ’d' +{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 5.4810665441672, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 73.0, "lr": 9e-06, "finish_rate": 0.781, "comp_len": 558.1, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 41.2, "frames": {"chat": 215}, "mem_gb": 10.0} +{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 5.1878687764207525, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 56.0, "lr": 1.2e-05, "finish_rate": 0.825, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.6, "frames": {"chat": 217}, "mem_gb": 9.88} +{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 4.475190736228227, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 48.5, "lr": 1.5e-05, "finish_rate": 0.8, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.4, "frames": {"chat": 205}, "mem_gb": 9.94} +{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 3.5818144115626813, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 44.5, "lr": 1.8e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 229}, "mem_gb": 9.91} +{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 3.2216264387488365, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 29.75, "lr": 2.1e-05, "finish_rate": 0.812, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 223}, "mem_gb": 9.98} +{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 2.623535114067793, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 28.75, "lr": 2.4e-05, "finish_rate": 0.708, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 202}, "mem_gb": 10.02} +{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 2.0890392102817694, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 21.375, "lr": 2.7000000000000002e-05, "finish_rate": 0.77, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 209}, "mem_gb": 9.99} +{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.5289459430545569, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 12.3125, "lr": 3e-05, "finish_rate": 0.885, "comp_len": 528.6, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 227}, "mem_gb": 9.96} +{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.3420937741284569, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 18.375, "lr": 3e-05, "finish_rate": 0.848, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 230}, "mem_gb": 10.04} +[eval step 10] sample: "To solve the problem, we need to determine the possible values of \\(a\\), \\(b\\), and \\(c\\) that satisfy the given equation \\(a + b + m + r = 18\\).\n\nLet's break down the problem into the steps:\n\n1. **Un" +{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.0470961780856054, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 8.5, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 231}, "mem_gb": 9.89} +{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8973094273423156, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 5.5625, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 245}, "mem_gb": 9.96} +{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7729372168297569, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 3.1875, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.7, "frames": {"chat": 210}, "mem_gb": 9.97} +{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7965288292273879, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 3.171875, "lr": 3e-05, "finish_rate": 0.758, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 211}, "mem_gb": 9.97} +{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5997185650259257, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 2.03125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 221}, "mem_gb": 10.02} +{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5663720039173961, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 1.8359375, "lr": 3e-05, "finish_rate": 0.912, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 250}, "mem_gb": 9.84} +{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5484086360000074, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 1.703125, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 229}, "mem_gb": 10.01} +{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4377009954464932, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 1.2421875, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 250}, "mem_gb": 9.99} +{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.47375443490048247, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 1.28125, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 231}, "mem_gb": 9.86} +{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4232544344509641, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 1.09375, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 224}, "mem_gb": 9.9} +[eval step 20] sample: "To solve the problem, we need to determine the values of \\(a\\), \\(b\\), and \\(m\\) such that the given equations are satisfied.\n\nLet's break down the problem step-by-step:\n\n1. **Understand the Equations" +{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.38542554356232284, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.98828125, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 212}, "mem_gb": 9.95} +{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.34992314758387705, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.91796875, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 238}, "mem_gb": 9.9} +{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.37193605013415215, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 1.0078125, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 257}, "mem_gb": 9.78} +{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32420069123518963, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.796875, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 227}, "mem_gb": 9.97} +{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.36334449760143955, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.83203125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 228}, "mem_gb": 10.0} +{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3422913500872751, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.89453125, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3060648350310822, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.71484375, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4031517359685153, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 26.5, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 197}, "mem_gb": 10.08} +{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3455044850113491, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.80859375, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 239}, "mem_gb": 9.83} +{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3500898130904883, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.8828125, "lr": 3e-05, "finish_rate": 0.83, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 224}, "mem_gb": 9.88} +[eval step 30] sample: 'To solve the problem, we need to determine the values of \\(a\\), \\(b\\), and \\(p\\) that satisfy the given equations:\n\n1. \\(a + b = k\\)\n2. \\(k + m = p\\)\n3. \\(p + a = r\\)\n4.' +{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2807131997364263, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 217}, "mem_gb": 9.99} +{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27056356236139933, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.703125, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 241}, "mem_gb": 9.99} +{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28164530578665437, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.68359375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 218}, "mem_gb": 9.97} +{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2762345147125423, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.64453125, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 206}, "mem_gb": 9.98} +{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2955992032624781, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.71484375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 232}, "mem_gb": 10.02} +{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.29579542716468377, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.69140625, "lr": 3e-05, "finish_rate": 0.771, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 218}, "mem_gb": 10.04} +{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28874229360707104, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.63671875, "lr": 3e-05, "finish_rate": 0.779, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 213}, "mem_gb": 10.0} +{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2895699510930727, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.70703125, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 248}, "mem_gb": 9.97} +{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2556835312332958, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 218}, "mem_gb": 10.03} +{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2271388866957277, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.851, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 221}, "mem_gb": 9.98} +[eval step 40] sample: 'To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), and \\(p\\) that satisfy the given equations:\n\n1. \\(a + b = k\\)\n2. \\(k + m = p\\)\n3. \\(p + a = r' +{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2487573687321196, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.59765625, "lr": 3e-05, "finish_rate": 0.894, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 236}, "mem_gb": 9.92} +{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2528259696563085, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 246}, "mem_gb": 9.84} +{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24962496552144486, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 234}, "mem_gb": 10.09} +{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20244326410479843, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.748, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 202}, "mem_gb": 9.97} +{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2265090825246026, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 217}, "mem_gb": 9.99} +{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23977290795333683, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.866, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 224}, "mem_gb": 9.99} +{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.26650513206385074, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.59765625, "lr": 3e-05, "finish_rate": 0.753, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 215}, "mem_gb": 10.0} +{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.216942871193029, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 1.2265625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 259}, "mem_gb": 9.92} +{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24050438385202239, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.59765625, "lr": 3e-05, "finish_rate": 0.829, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 210}, "mem_gb": 9.99} +{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2729782617907971, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.59375, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 213}, "mem_gb": 10.04} +[eval step 50] sample: 'To solve the given system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), and \\(p\\) such that:\n\n\\[\na + b = k\n\\]\n\\[\nk + m = p\n\\]\n\\[\np + a = r\n' +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep25_s1224/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23127874396915238, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.0, "frames": {"chat": 222}, "mem_gb": 9.95} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2022097536571324, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 235}, "mem_gb": 10.0} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22392032471460602, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 208}, "mem_gb": 9.96} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21173793127896884, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 191}, "mem_gb": 10.0} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1773922070093453, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 219}, "mem_gb": 9.99} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1603636117445305, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 207}, "mem_gb": 10.0} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24646623886699479, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 208}, "mem_gb": 9.95} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16445816849029313, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 219}, "mem_gb": 9.99} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18552401562519372, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 246}, "mem_gb": 9.87} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23117222544706117, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 206}, "mem_gb": 10.02} +[eval step 60] sample: 'To solve the given system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), and \\(p\\) such that:\n\n\\[\na + b = k\n\\]\n\\[\nk + m = p\n\\]\n\\[\np + a = r\n' +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18189995611303797, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 233}, "mem_gb": 10.0} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18372100273153436, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.8671875, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 229}, "mem_gb": 9.86} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1487495786678667, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 236}, "mem_gb": 9.9} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18451065494281552, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 239}, "mem_gb": 9.78} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16504728367341062, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 241}, "mem_gb": 9.9} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18468811580395947, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 226}, "mem_gb": 9.87} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15604329239850243, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 234}, "mem_gb": 10.0} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16107819736519208, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 257}, "mem_gb": 9.99} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23071295691871394, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.52734375, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 208}, "mem_gb": 10.04} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19346359119676054, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 211}, "mem_gb": 10.02} +[eval step 70] sample: 'To solve the given system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), and \\(p\\) such that:\n\n\\[\na + b = k\n\\]\n\\[\nk + m = p\n\\]\n\\[\np + a = r\n' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20979610983723154, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 227}, "mem_gb": 10.0} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18511814717464148, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 211}, "mem_gb": 9.98} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15541142247617246, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 238}, "mem_gb": 9.99} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16142152046722671, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 237}, "mem_gb": 10.03} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18345416961008063, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 208}, "mem_gb": 10.03} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16612481657912334, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 221}, "mem_gb": 10.12} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18476293021012097, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.52734375, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 232}, "mem_gb": 9.96} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1682081102202336, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 208}, "mem_gb": 9.99} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16280597681732228, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 227}, "mem_gb": 9.91} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18280604339229564, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 221}, "mem_gb": 9.94} +[eval step 80] sample: 'To solve the given system of equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe can follow these steps:\n\n' +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14743654915646962, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 232}, "mem_gb": 10.0} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15533407865427434, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17884458175196002, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 195}, "mem_gb": 10.09} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17335635773831357, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 216}, "mem_gb": 10.0} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15658746059002976, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 208}, "mem_gb": 9.88} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15384809637721628, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.5, "frames": {"chat": 235}, "mem_gb": 9.88} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17239943368534247, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 225}, "mem_gb": 9.99} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20018711051177235, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 213}, "mem_gb": 10.08} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1512549991114065, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 257}, "mem_gb": 9.75} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19405477244090288, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 212}, "mem_gb": 10.02} +[eval step 90] sample: 'To solve the given system of equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe can follow these steps:\n\n' +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16499492380482456, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 221}, "mem_gb": 10.0} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15074481266401707, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 242}, "mem_gb": 9.99} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12890028902329503, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 220}, "mem_gb": 9.96} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1444535345125012, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 240}, "mem_gb": 9.85} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15116780090536922, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 206}, "mem_gb": 9.98} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16743213449095687, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 226}, "mem_gb": 10.0} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21188729687035085, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 244}, "mem_gb": 9.78} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20142290921670694, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 224}, "mem_gb": 10.0} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1621253117416054, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 271}, "mem_gb": 9.72} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15016368210408837, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 236}, "mem_gb": 10.01} +[eval step 100] sample: 'To solve this problem, we need to translate the given equations into a system of linear equations and solve for the unknowns \\(a\\), \\(b\\), \\(m\\), and \\(p\\).\n\nGiven:\n\\[\n\\begin{align*}\na + b &= k \\\\\nk +' +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep25_s1224/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1557210696899332, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 232}, "mem_gb": 9.88} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1422160825269297, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.6, "frames": {"chat": 210}, "mem_gb": 9.93} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1296106172949386, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.2, "t_rollout_s": 0.0, "t_step_s": 41.4, "frames": {"chat": 217}, "mem_gb": 9.9} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16544860142044102, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 224}, "mem_gb": 10.02} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20553272317995627, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 203}, "mem_gb": 9.87} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15951158448743324, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 239}, "mem_gb": 9.97} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11594521766655767, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 254}, "mem_gb": 9.88} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10944677127550045, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 241}, "mem_gb": 9.97} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17688703423095867, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 213}, "mem_gb": 10.0} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13732703933753074, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 221}, "mem_gb": 10.05} +[eval step 110] sample: 'To solve the given system of equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe can follow these steps:\n\n' +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.153877969346568, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.4, "frames": {"chat": 196}, "mem_gb": 10.01} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1227314798528639, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 270}, "mem_gb": 9.81} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1115176025235715, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 216}, "mem_gb": 9.99} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1313754518270803, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.4, "frames": {"chat": 200}, "mem_gb": 9.96} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1082449041414385, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 206}, "mem_gb": 9.91} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10967711655922854, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 234}, "mem_gb": 9.94} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1271924709101518, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 215}, "mem_gb": 9.95} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11125657115193704, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 255}, "mem_gb": 9.94} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12195842446442694, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 250}, "mem_gb": 9.82} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1218588234690018, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 242}, "mem_gb": 9.99} +[eval step 120] sample: 'To solve the system of equations given:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe can follow these steps:\n\n' +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13956055214075994, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1709896477163459, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 208}, "mem_gb": 10.03} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11171058443682268, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 208}, "mem_gb": 9.97} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13555439033694566, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 209}, "mem_gb": 10.12} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10593995610407243, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 235}, "mem_gb": 9.95} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11018521934015056, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 204}, "mem_gb": 9.94} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1523059667794034, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 208}, "mem_gb": 10.0} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11395702697544668, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 240}, "mem_gb": 10.0} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10960895141505947, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 246}, "mem_gb": 9.99} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12840980574265123, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 243}, "mem_gb": 9.81} +[eval step 130] sample: 'To solve the given system of equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe need to determine the values of \\(' +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1445476036771511, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.4, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12889366007174055, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14989680531652022, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 211}, "mem_gb": 10.01} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12166979752990106, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 232}, "mem_gb": 9.97} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1422418523649064, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 214}, "mem_gb": 10.0} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1125547682431837, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 226}, "mem_gb": 9.89} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11254967547400545, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10425090258192891, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 218}, "mem_gb": 9.83} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10396384794348851, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 233}, "mem_gb": 9.98} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1503090637408197, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 215}, "mem_gb": 10.0} +[eval step 140] sample: 'To solve the given system of equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe can follow these steps:\n\n' +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12397584599244098, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1101376080014122, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 209}, "mem_gb": 9.94} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10867827801605065, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 262}, "mem_gb": 9.87} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11113408046951517, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 249}, "mem_gb": 9.96} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1363239465509231, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 227}, "mem_gb": 9.99} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.102995293696442, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 221}, "mem_gb": 9.99} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10967131499343862, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 234}, "mem_gb": 10.01} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10105065601477399, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 213}, "mem_gb": 9.95} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10403523254335548, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 213}, "mem_gb": 9.89} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10564743132308746, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 234}, "mem_gb": 9.92} +[eval step 150] sample: "To solve the given system of equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe'll follow these steps:\n\n" +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep25_s1224/step0150 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: uploading summary, console lines 169-170 +wandb: +wandb: Run history: +wandb: comp_len ▇▄▆▅▄▂█▃▆▆▃▅▆▆▆▄▇▅▅▇▅▄▆▂▆▇▅▅▁▆▅▆▅▁▆▃▃▆▄▄ +wandb: cumulative_loss_tokens ▁▁▁▂▂▂▂▂▂▂▃▃▃▃▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▅▅▆▆▆▆▆▇▇█ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅██████████ +wandb: finish_rate ▅▃▅▂▄▅▄▆▇▄▅▃▇▇▅▆▅▁▅▆▄▄▁▂▅█▄▅▄▇▂▇▂▂▂▆▃█▇▄ +wandb: forward_topk_kl █▂▂▂▂▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm █▆▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▇█████████████████████████████████████ +wandb: mem_gb ▃▆▄▅▂▅▅▅▅▅▆▃▇▅▅▅▂▆▂▃▆▅▃▄▄▆▂▅▆▅▆█▄▅▁▃▆▅▅▃ +wandb: step ▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▅▅▅▅▆▆▆▆▆▆▆▆▇▇▇▇▇▇▇██ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 512.8 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.906 +wandb: forward_topk_kl 0.10565 +wandb: grad_norm 0.34961 +wandb: lr 3e-05 +wandb: mem_gb 9.92 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run reap-math-keep25-s1224 at: https://wandb.ai/hbfreed/glean-grid/runs/q0iymgs1 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/reap_keep25_s1224/wandb/run-20260716_071102-q0iymgs1/logs +{ + "correct": 134, + "accuracy": 0.10159211523881728, + "finished": 1084, + "finish_rate": 0.8218347232752085, + "mean_completion_tokens": 193.42304776345716 +} +saved item-level results -> outputs/evals/grid_math/reap_keep25_s1224_step100_chat.json +{ + "correct": 156, + "accuracy": 0.11827141774071266, + "finished": 1076, + "finish_rate": 0.8157695223654283, + "mean_completion_tokens": 191.31084154662622 +} +saved item-level results -> outputs/evals/grid_math/reap_keep25_s1224_step150_chat.json diff --git a/healed/grid_math/reap_keep25_s1226.console.log b/healed/grid_math/reap_keep25_s1226.console.log new file mode 100644 index 0000000000000000000000000000000000000000..57181ae78014b4506ff18ff7123e35520b618e86 --- /dev/null +++ b/healed/grid_math/reap_keep25_s1226.console.log @@ -0,0 +1,231 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run ep759fa6 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/reap_keep25_s1226/wandb/run-20260716_063351-ep759fa6 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run reap-math-keep25-s1226 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/ep759fa6 +12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 2.09B | teacher overlap=False +{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 5.3948620602846145, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 83.5, "lr": 6e-06, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 254}, "mem_gb": 9.81} +The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results. +[eval step 1] sample: '| 16:\nSnrd.\n/VA_#(t_i\n\nTo break n-day::a\n\nas\n\nisntABent\n\nThe a * a bente\n\n|A2 a\n| a.b,b\nfor [\n{U' +{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 5.492758668692907, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 68.5, "lr": 9e-06, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 241}, "mem_gb": 9.97} +{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 5.288580652968089, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 50.25, "lr": 1.2e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 213}, "mem_gb": 10.0} +{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 4.361503725457191, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 47.75, "lr": 1.5e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.7, "frames": {"chat": 221}, "mem_gb": 10.05} +{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 3.849147217363119, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 39.5, "lr": 1.8e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.9, "frames": {"chat": 196}, "mem_gb": 10.01} +{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 3.022541018138329, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 28.0, "lr": 2.1e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 270}, "mem_gb": 9.81} +{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 2.5903835578064123, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 36.5, "lr": 2.4e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.2, "frames": {"chat": 216}, "mem_gb": 9.99} +{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.989209920862317, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 22.375, "lr": 2.7000000000000002e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.3, "frames": {"chat": 200}, "mem_gb": 9.96} +{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.5486149408777554, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 13.375, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.5, "frames": {"chat": 206}, "mem_gb": 9.91} +{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.1493749193057419, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 12.875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 234}, "mem_gb": 9.94} +[eval step 10] sample: 'To solve the problem, we need to determine the perimeter of the triangle. We are given the the perimeter of the triangle is 28. The perimeter of a triangle is the sum of the distances of the sides.\n\nL' +{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.0132465209826826, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 8.375, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.0, "frames": {"chat": 215}, "mem_gb": 9.95} +{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.79206941087991, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 4.59375, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 255}, "mem_gb": 9.94} +{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7040475417902072, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 7.375, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 250}, "mem_gb": 9.82} +{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6600783367355665, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 10.9375, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 242}, "mem_gb": 9.99} +{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6760948089194795, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 8.0, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.735672302259008, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 4.78125, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 208}, "mem_gb": 10.03} +{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5825303116974732, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 4.6875, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.0, "frames": {"chat": 208}, "mem_gb": 9.97} +{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6073473408266902, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 2.203125, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 209}, "mem_gb": 10.12} +{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.45263020926391084, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 1.359375, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 235}, "mem_gb": 9.95} +{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4413063018143177, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 3.25, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.4, "frames": {"chat": 204}, "mem_gb": 9.94} +[eval step 20] sample: 'To solve this problem, we need to determine the perimeter of the resulting triangle given the conditions:\n\n1. The perimeter of the triangle is 28.\n2. The midpoints of the sides of the triangle are con' +{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5014240723443528, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 1.453125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 208}, "mem_gb": 10.0} +{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.39647342192269863, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 1.0234375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 240}, "mem_gb": 10.0} +{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3701785091145585, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 1.03125, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 246}, "mem_gb": 9.99} +{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.38433438787100216, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 1.03125, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.7, "frames": {"chat": 243}, "mem_gb": 9.81} +{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.39559759280495344, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 1.2109375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.4, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.35509972168381015, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 1.3203125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.4, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.39220244832945367, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.91796875, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 211}, "mem_gb": 10.01} +{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32253637241298955, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.83203125, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 232}, "mem_gb": 9.97} +{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3454374578539282, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.84375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.4, "frames": {"chat": 214}, "mem_gb": 10.0} +{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.30258261669923864, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.73828125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 226}, "mem_gb": 9.89} +[eval step 30] sample: 'To solve this problem, we need to determine the perimeter of the resulting triangle formed by connecting the midpoints of the sides of the given triangle.\n\nHere are the steps to solve the problem:\n\n1.' +{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28547664239617687, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.66015625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2539228688845411, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.65234375, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.4, "frames": {"chat": 218}, "mem_gb": 9.83} +{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.26798752832549316, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.6640625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 233}, "mem_gb": 9.98} +{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3186436773126324, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.69921875, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 215}, "mem_gb": 10.0} +{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2920498409892122, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.4, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2473691465223829, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 209}, "mem_gb": 9.94} +{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25275046684630215, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 262}, "mem_gb": 9.87} +{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2435881425558279, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.57421875, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 249}, "mem_gb": 9.96} +{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.29404913728336496, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 227}, "mem_gb": 9.99} +{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22125561543889344, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 221}, "mem_gb": 9.99} +[eval step 40] sample: 'To solve this problem, we need to understand the geometric properties and the relationships between the sides of the triangle formed by the midpoints of the original triangle.\n\nGiven:\n- The perimeter ' +{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2520220772454515, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 234}, "mem_gb": 10.01} +{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20081105666371682, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.1, "frames": {"chat": 213}, "mem_gb": 9.95} +{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2016371044045935, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.4, "frames": {"chat": 213}, "mem_gb": 9.89} +{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22321325843880574, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.54296875, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 234}, "mem_gb": 9.92} +{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21784419940554847, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 222}, "mem_gb": 9.99} +{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2611530060334752, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.59375, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 227}, "mem_gb": 10.0} +{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22217773687231043, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.4, "frames": {"chat": 218}, "mem_gb": 10.04} +{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2560874344268193, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.66796875, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 223}, "mem_gb": 10.01} +{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2610147045496851, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.546875, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 206}, "mem_gb": 10.0} +{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22958950562837224, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 213}, "mem_gb": 9.92} +[eval step 50] sample: "To solve this problem, we need to understand the geometric properties and the given conditions. Here's a step-by-step approach:\n\n1. **Understand the Problem:**\n - The perimeter of the triangle is 28" +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep25_s1226/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2759530112889906, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.5859375, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 223}, "mem_gb": 9.86} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24462636410264918, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 227}, "mem_gb": 9.97} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2153538477362444, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 253}, "mem_gb": 10.0} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2091132113194093, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.8, "frames": {"chat": 216}, "mem_gb": 10.0} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2178824202261865, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 205}, "mem_gb": 9.97} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2252128013719494, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 207}, "mem_gb": 10.06} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23820067919362337, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 215}, "mem_gb": 9.98} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1530524639653663, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 228}, "mem_gb": 10.0} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2108373320500677, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.54296875, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 221}, "mem_gb": 10.04} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16624215446698168, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 254}, "mem_gb": 9.84} +[eval step 60] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and how the midpoints of its sides affect the perimeter.\n\n### Step-by-Step Solution:\n\n1. **Understand the Geometry' +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14189707856637737, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.0, "frames": {"chat": 210}, "mem_gb": 9.96} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16133386867145696, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 226}, "mem_gb": 9.92} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.184017719945622, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.439453125, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 212}, "mem_gb": 9.99} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2009096172561869, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 211}, "mem_gb": 9.92} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2035785714487235, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.0, "frames": {"chat": 196}, "mem_gb": 9.97} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15334912759084254, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 212}, "mem_gb": 9.99} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16262201901270698, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 244}, "mem_gb": 9.91} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14996965130375078, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 222}, "mem_gb": 9.95} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.182706143669722, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 218}, "mem_gb": 10.0} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17859229601956905, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 252}, "mem_gb": 9.87} +[eval step 70] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and how the midpoints of its sides are connected by segments.\n\n### Step-by-Step Solution:\n\n1. **Understand the Geo' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19512056777638695, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 202}, "mem_gb": 10.05} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22794473583027722, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 237}, "mem_gb": 10.0} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1737093073921899, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 234}, "mem_gb": 9.98} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13047870258555438, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 215}, "mem_gb": 10.0} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13498239639320722, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 234}, "mem_gb": 9.93} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12114300584094599, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 216}, "mem_gb": 9.98} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1555439574915295, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.3, "frames": {"chat": 210}, "mem_gb": 9.95} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1471449433473715, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.5, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1342087133670226, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.3, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1672618577302744, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 225}, "mem_gb": 9.95} +[eval step 80] sample: 'To solve this problem, we need to understand the geometric properties and the relationships between the sides of the triangle formed by the midpoints of the original triangle.\n\n### Steps to Solve the ' +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1618539959486574, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 253}, "mem_gb": 9.85} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15880980721451343, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 247}, "mem_gb": 9.97} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15618948619918277, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 238}, "mem_gb": 9.97} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19067880896230538, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 235}, "mem_gb": 9.99} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18032733394745737, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 215}, "mem_gb": 9.96} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1783826366210977, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 268}, "mem_gb": 9.97} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17707187270099917, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 228}, "mem_gb": 10.0} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1674367210490629, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 252}, "mem_gb": 9.93} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1522629966557026, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 223}, "mem_gb": 10.01} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2198721208633855, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.6171875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 226}, "mem_gb": 9.99} +[eval step 90] sample: 'To solve this problem, we need to understand the geometric properties and the relationships between the sides of the triangle formed by the midpoints of the original triangle.\n\n### Steps to Solve:\n\n1.' +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19896256684847177, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 208}, "mem_gb": 10.04} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14341324961800128, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 240}, "mem_gb": 9.93} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16562857267570993, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 222}, "mem_gb": 9.92} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15155938894315624, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 236}, "mem_gb": 9.99} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12592543063924339, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.2, "frames": {"chat": 217}, "mem_gb": 9.96} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15650399845099697, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 252}, "mem_gb": 9.87} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13744228080461424, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 222}, "mem_gb": 9.99} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14708479140317068, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.901, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 242}, "mem_gb": 9.87} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19175601966977118, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 219}, "mem_gb": 9.93} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14486887626430642, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 223}, "mem_gb": 9.94} +[eval step 100] sample: "To solve this problem, we need to understand the geometric properties and the given conditions. Here's how we can break it down:\n\n1. **Understand the Problem:**\n - We have a triangle with a perimete" +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep25_s1226/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17521546159796417, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 232}, "mem_gb": 9.94} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17657650477966916, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.832, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 220}, "mem_gb": 10.0} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1642237206614266, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 210}, "mem_gb": 10.04} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14797473591733723, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 226}, "mem_gb": 9.97} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13756006544300667, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.741, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 212}, "mem_gb": 9.99} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14728604319685448, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 236}, "mem_gb": 10.01} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11492285058296596, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.928, "comp_len": 454.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 264}, "mem_gb": 9.88} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16432044878825544, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 229}, "mem_gb": 9.98} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11607246005069465, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 465.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 258}, "mem_gb": 9.85} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16434998966678976, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 208}, "mem_gb": 10.01} +[eval step 110] sample: 'To solve this problem, we need to understand the geometric properties and the relationships between the sides of the triangle and the midpoints.\n\nGiven:\n- The perimeter of the triangle is 28.\n- The mi' +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12820218563723998, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.88, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 249}, "mem_gb": 9.92} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10775780974778656, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.5, "frames": {"chat": 220}, "mem_gb": 9.99} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13256505108779917, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.9, "frames": {"chat": 223}, "mem_gb": 9.98} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.107142218930802, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 221}, "mem_gb": 9.99} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10337071840027347, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 230}, "mem_gb": 9.9} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12823014822658152, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 214}, "mem_gb": 9.97} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17329245272489885, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 214}, "mem_gb": 9.99} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12728931551845743, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 210}, "mem_gb": 10.03} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13437456069858744, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 214}, "mem_gb": 9.99} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1287463510526344, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.791, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 215}, "mem_gb": 9.95} +[eval step 120] sample: "To solve this problem, we need to understand the geometric properties and relationships involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the triangle is" +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1430338466726554, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 208}, "mem_gb": 9.99} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10728711698964859, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.789, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 218}, "mem_gb": 9.87} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09843137963234136, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 233}, "mem_gb": 9.89} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09210390640692785, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 231}, "mem_gb": 9.93} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12565180675198015, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 235}, "mem_gb": 10.13} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12365243875219797, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.3, "frames": {"chat": 216}, "mem_gb": 9.98} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11053694897955284, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.831, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 237}, "mem_gb": 10.01} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15008561470514784, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 203}, "mem_gb": 10.01} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11617295994066323, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.873, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 236}, "mem_gb": 10.03} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11612723016667491, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 214}, "mem_gb": 10.0} +[eval step 130] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and how its midpoints are connected.\n\n### Steps to Solve the Problem:\n\n1. **Understand the Geometry:**\n - Let th' +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11800649179657921, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 209}, "mem_gb": 9.99} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11760378193684543, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 256}, "mem_gb": 10.0} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1103896147995256, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 230}, "mem_gb": 9.86} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11834490465267251, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 230}, "mem_gb": 10.05} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13661176628569763, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 227}, "mem_gb": 9.95} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1438256380248815, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13295528639039644, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.699, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 206}, "mem_gb": 10.03} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13133824970157196, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.82, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 228}, "mem_gb": 9.9} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12861836654854317, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 224}, "mem_gb": 9.99} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1111085697865424, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.66, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 200}, "mem_gb": 10.03} +[eval step 140] sample: "To solve this problem, we need to understand the geometric properties and relationships involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - We have a triangle with a perime" +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12692416681302712, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.714, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.8, "frames": {"chat": 196}, "mem_gb": 10.01} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10248101542762791, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 223}, "mem_gb": 9.99} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10177099369396456, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.869, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.6, "frames": {"chat": 213}, "mem_gb": 9.88} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10673611074338357, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 232}, "mem_gb": 9.93} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10685085613004243, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 223}, "mem_gb": 9.92} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11504278169833124, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.85, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 233}, "mem_gb": 10.02} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11391995535188665, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.816, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.3, "frames": {"chat": 217}, "mem_gb": 10.01} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1639873969303444, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.752, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 202}, "mem_gb": 10.07} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10495261400962869, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 253}, "mem_gb": 9.93} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0964433067208156, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 231}, "mem_gb": 9.94} +[eval step 150] sample: "To solve this problem, we need to understand the geometric properties and relationships involved. Here's a step-by-step approach:\n\n1. **Understand the Problem:**\n - We have a triangle with a perimet" +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep25_s1226/step0150 +wandb: updating run metadata +wandb: uploading summary, console lines 170-170 +wandb: +wandb: Run history: +wandb: comp_len ▆▁▆▃▃▇▇▇▃▅▆▇▅▄▇▅▇▂█▆█▄▄▂▁▅▅▄▅▄▁▃▆▅▅▆▆▄▂▄ +wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▆▇▇▇▇█ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅███████████ +wandb: finish_rate ▂▆█▄▅▂▇▂▄▆▇▄▄▃▅▃▆▅▂▆▃▁▅▇▄▅▇▇▆▂▇▆▅▃▃▆▂▄▆▂ +wandb: forward_topk_kl █▇▄▄▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm █▄▃▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▆▇████████████████████████████████████ +wandb: mem_gb █▁▅▇▆▇▇▇▇▇▆▅▇▆▇█▆▆▅▆▇▂▆▇▆▃▃▇▃▆▅▃▅▇█▇▆▃▇▅ +wandb: step ▁▁▂▂▂▂▂▂▂▂▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇▇▇██ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 519.5 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.879 +wandb: forward_topk_kl 0.09644 +wandb: grad_norm 0.32422 +wandb: lr 3e-05 +wandb: mem_gb 9.94 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run reap-math-keep25-s1226 at: https://wandb.ai/hbfreed/glean-grid/runs/ep759fa6 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/reap_keep25_s1226/wandb/run-20260716_063351-ep759fa6/logs +{ + "correct": 158, + "accuracy": 0.1197877179681577, + "finished": 923, + "finish_rate": 0.6997725549658832, + "mean_completion_tokens": 233.13343442001516 +} +saved item-level results -> outputs/evals/grid_math/reap_keep25_s1226_step100_chat.json +{ + "correct": 163, + "accuracy": 0.12357846853677028, + "finished": 1005, + "finish_rate": 0.7619408642911296, + "mean_completion_tokens": 214.75056861258528 +} +saved item-level results -> outputs/evals/grid_math/reap_keep25_s1226_step150_chat.json diff --git a/healed/grid_math/reap_keep50_s1224.console.log b/healed/grid_math/reap_keep50_s1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..633f4e55fa76c1afde2860b628e212a84238e05c --- /dev/null +++ b/healed/grid_math/reap_keep50_s1224.console.log @@ -0,0 +1,233 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run 1c8m5fh6 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/reap_keep50_s1224/wandb/run-20260716_015810-1c8m5fh6 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run reap-math-keep50-s1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/1c8m5fh6 + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/grid_math/reap_keep50_s1224/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08243085641801978, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 222}, "mem_gb": 16.0} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0740578943549345, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.3, "frames": {"chat": 235}, "mem_gb": 16.05} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08452489547633256, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 208}, "mem_gb": 16.01} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07001415301486849, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 191}, "mem_gb": 16.05} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0545745571903574, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 219}, "mem_gb": 16.04} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05674678605599329, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 207}, "mem_gb": 16.05} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07849393064022685, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 208}, "mem_gb": 16.0} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05212787776181164, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.2431640625, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 219}, "mem_gb": 16.04} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05130243318529489, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.8, "frames": {"chat": 246}, "mem_gb": 15.92} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07500699858136164, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 206}, "mem_gb": 16.07} +[eval step 60] sample: 'To solve the problem, we need to find the digits \\(a, b, k, m, r\\) such that each digit is a non-zero digit (i.e., between 1 and 9) and satisfies the given equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\' +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05669750292877046, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 233}, "mem_gb": 16.05} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05847539465183703, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 229}, "mem_gb": 15.91} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04844146788320504, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.2431640625, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 236}, "mem_gb": 15.95} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05839694087315972, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 239}, "mem_gb": 15.83} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.049815410684685535, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.240234375, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 241}, "mem_gb": 15.95} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05642013670879727, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 226}, "mem_gb": 15.92} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04739125943775289, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.2392578125, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 234}, "mem_gb": 16.05} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04787641853652894, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.23828125, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 257}, "mem_gb": 16.04} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07120997395546486, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 208}, "mem_gb": 16.09} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.063085703232248, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.3, "frames": {"chat": 211}, "mem_gb": 16.07} +[eval step 70] sample: 'To solve the problem, we need to find the digits \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that each letter represents a non-zero digit and satisfies the given equations:\n\n\\[\n\\begin{align*}\na + b &= ' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06971925978801834, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 227}, "mem_gb": 16.05} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.061769093114696444, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 211}, "mem_gb": 16.03} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05006525234617293, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 238}, "mem_gb": 16.04} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05073510147240013, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.2392578125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.5, "frames": {"chat": 237}, "mem_gb": 16.08} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0631699871141774, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 208}, "mem_gb": 16.08} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.051907375587942076, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.2451171875, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 221}, "mem_gb": 16.17} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.056417306477592015, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.76953125, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 232}, "mem_gb": 16.01} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05626846161660117, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 208}, "mem_gb": 16.04} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.049228766723753266, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 227}, "mem_gb": 15.96} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05534716739145418, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 221}, "mem_gb": 15.99} +[eval step 80] sample: "To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(p\\), and \\(r\\) such that each letter represents a non-zero digit and the given equations hold true.\n\nLet's break down th" +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.047819643541720386, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.2431640625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 232}, "mem_gb": 16.05} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.056753374835678064, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 219}, "mem_gb": 16.05} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05573002262612184, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.2490234375, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 195}, "mem_gb": 16.14} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05441774775936889, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 216}, "mem_gb": 16.05} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06264704996475484, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 208}, "mem_gb": 15.93} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05151226184510936, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 235}, "mem_gb": 15.93} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05486625511728538, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 225}, "mem_gb": 16.04} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06505359720261768, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 213}, "mem_gb": 16.13} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.044957342780055476, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.0, "frames": {"chat": 257}, "mem_gb": 15.8} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06199448381081844, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 212}, "mem_gb": 16.07} +[eval step 90] sample: 'To solve the problem, we need to find the digits \\(a, b, k, m, r\\) such that each digit is a non-zero digit (i.e., 1-9) and satisfies the given equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\n' +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05085727345328778, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 221}, "mem_gb": 16.05} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.046645166625843074, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.2333984375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 242}, "mem_gb": 16.04} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.046637191681253416, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 220}, "mem_gb": 16.01} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.044484876428404825, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 240}, "mem_gb": 15.9} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05077935193019609, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 206}, "mem_gb": 16.03} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05764281274767903, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 226}, "mem_gb": 16.05} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07217985147928509, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 244}, "mem_gb": 15.83} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.062065851740368334, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.9, "frames": {"chat": 224}, "mem_gb": 16.05} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04666607260017966, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 271}, "mem_gb": 15.77} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05521788966048819, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 236}, "mem_gb": 16.06} +[eval step 100] sample: 'To solve this problem, we need to find the digits \\(a, b, k, m, r\\) such that each letter represents a non-zero digit and satisfy the given equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np ' +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep50_s1224/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04969318243367597, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 232}, "mem_gb": 15.93} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05082408647788689, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 210}, "mem_gb": 15.98} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05025460629562537, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 217}, "mem_gb": 15.95} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05321642030389048, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.0, "frames": {"chat": 224}, "mem_gb": 16.07} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06337893520779908, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 203}, "mem_gb": 15.92} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.050805029303006205, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.5, "frames": {"chat": 239}, "mem_gb": 16.02} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.033205773511280616, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 254}, "mem_gb": 15.93} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.033417627344016605, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.2353515625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 241}, "mem_gb": 16.02} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04776025845406887, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.216796875, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 213}, "mem_gb": 16.05} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04037379693359447, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.2119140625, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 221}, "mem_gb": 16.1} +[eval step 110] sample: 'To solve the problem, we need to find the digits \\(a, b, k, m, r\\) such that each digit is a non-zero digit (i.e., between 1 and 9) and satisfies the given equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\' +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.044434299368970094, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.2412109375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 196}, "mem_gb": 16.06} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0333152589352103, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.2080078125, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.0, "frames": {"chat": 270}, "mem_gb": 15.86} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03534065244613836, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.21484375, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 216}, "mem_gb": 16.04} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.040529147774881376, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.197265625, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 200}, "mem_gb": 16.01} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03757092032081758, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 206}, "mem_gb": 15.96} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03247500213080396, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.1923828125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 234}, "mem_gb": 15.99} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.037848325089799864, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.2109375, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 215}, "mem_gb": 16.0} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03213813143230509, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 255}, "mem_gb": 15.99} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.032368304668770484, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.1982421875, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 250}, "mem_gb": 15.87} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.033178579137373404, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.197265625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 242}, "mem_gb": 16.04} +[eval step 120] sample: "To solve the problem, we need to find the digits \\(a, b, k, m, p,\\) and \\(r\\) such that each letter represents a non-zero digit and the given equations hold true. Let's break down the problem step-by-" +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04187165923851232, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.2177734375, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 199}, "mem_gb": 16.05} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.049618725496840974, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 208}, "mem_gb": 16.08} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03635383392153308, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 208}, "mem_gb": 16.02} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.043045021270743264, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.21875, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.2, "frames": {"chat": 209}, "mem_gb": 16.17} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.031225860221870242, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 235}, "mem_gb": 16.0} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.035118358117295426, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 204}, "mem_gb": 15.99} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.045566377886307116, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 208}, "mem_gb": 16.05} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.033411375429985735, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.2001953125, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 240}, "mem_gb": 16.05} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03442209031227976, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.2255859375, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 246}, "mem_gb": 16.04} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03496374331774811, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.1943359375, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 243}, "mem_gb": 15.86} +[eval step 130] sample: "To solve the given system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(r\\) such that each letter represents a non-zero digit. Let's break down the problem step-by" +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04256709007731018, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.2236328125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 208}, "mem_gb": 16.06} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04353734484561719, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.2216796875, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 219}, "mem_gb": 16.05} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.045320879577457285, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.2177734375, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 211}, "mem_gb": 16.06} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03994098002462027, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.2314453125, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 232}, "mem_gb": 16.02} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04629031352402332, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.224609375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 214}, "mem_gb": 16.05} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.036389342272576564, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 226}, "mem_gb": 15.94} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03558888955223374, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 210}, "mem_gb": 16.06} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03412930506222571, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 218}, "mem_gb": 15.88} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03288377221804112, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 233}, "mem_gb": 16.03} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04284478505373312, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.2119140625, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 215}, "mem_gb": 16.05} +[eval step 140] sample: 'To solve the problem, we need to find the digits \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that each letter represents a non-zero digit and satisfy the given equations:\n\n\\[\n\\begin{align*}\na + b &= k ' +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03632894418287712, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.19921875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 233}, "mem_gb": 16.04} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.035847735164438684, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.2294921875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 209}, "mem_gb": 15.99} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.029498823017751176, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.1708984375, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 262}, "mem_gb": 15.92} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03295737003815205, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 249}, "mem_gb": 16.01} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.041888812201377, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.19921875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 227}, "mem_gb": 16.04} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03317792658337858, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.1875, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 221}, "mem_gb": 16.04} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03243056264965174, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.1875, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 234}, "mem_gb": 16.06} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03404021299170951, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.21484375, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 213}, "mem_gb": 16.0} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03388322359360754, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.1923828125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 213}, "mem_gb": 15.94} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.032232557944192865, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 234}, "mem_gb": 15.97} +[eval step 150] sample: "To solve the problem, we need to find the digits \\(a, b, k, m, r\\) such that each letter represents a non-zero digit and the given equations hold true. Let's break down the problem step-by-step:\n\n1. *" +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep50_s1224/step0150 +wandb: updating run metadata +wandb: uploading summary, console lines 171-171; uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: uploading data +wandb: +wandb: Run history: +wandb: comp_len ▃▆▄▄▆▄▄▃▄▃█▃▄▆▂▆▃▅▂▇▄▁▃▄▅▆▄▃▃▄▅█▁▆▆▃▂▆▄▆ +wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▂▂▃▃▃▄▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇████ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅█████████████ +wandb: finish_rate ▄▅▆▆▂▅▆▄▅▅▅▂▄▇▁▆▆▆▇▂▂▅▄▅█▅▆▅▁▄▄█▂█▁▂▇▄▂▇ +wandb: forward_topk_kl █▆▇▄▄▃▂▂▂▂▂▂▂▂▂▂▁▁▁▁▁▂▂▁▁▂▁▁▂▁▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm █▆▃▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▃▅████████████████████████████████████ +wandb: mem_gb ▁▄▇▅▂█▇▆▅▇▆▇▆▇▄▇▇█▇▇▆▂▄▅▅▆▄▆█▇▆▇▃▇▆▄▆▇▆▅ +wandb: step ▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▆▇▇▇▇▇▇▇██ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 512.8 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.906 +wandb: forward_topk_kl 0.03223 +wandb: grad_norm 0.18555 +wandb: lr 3e-05 +wandb: mem_gb 15.97 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run reap-math-keep50-s1224 at: https://wandb.ai/hbfreed/glean-grid/runs/1c8m5fh6 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/reap_keep50_s1224/wandb/run-20260716_015810-1c8m5fh6/logs +{ + "correct": 748, + "accuracy": 0.5670962850644428, + "finished": 1310, + "finish_rate": 0.9931766489764974, + "mean_completion_tokens": 121.06141015921152 +} +saved item-level results -> outputs/evals/grid_math/reap_keep50_s1224_step100_chat.json +{ + "correct": 777, + "accuracy": 0.5890826383623957, + "finished": 1307, + "finish_rate": 0.9909021986353298, + "mean_completion_tokens": 120.55724033358605 +} +saved item-level results -> outputs/evals/grid_math/reap_keep50_s1224_step150_chat.json diff --git a/healed/grid_math/reap_keep50_s1225.console.log b/healed/grid_math/reap_keep50_s1225.console.log new file mode 100644 index 0000000000000000000000000000000000000000..3f58365397b0dab6fab7f45ba3410c3adca83598 --- /dev/null +++ b/healed/grid_math/reap_keep50_s1225.console.log @@ -0,0 +1,232 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run 3wq993z8 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/reap_keep50_s1225/wandb/run-20260716_014634-3wq993z8 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run reap-math-keep50-s1225 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/3wq993z8 + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/grid_math/reap_keep50_s1225/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.09199335136047254, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 224}, "mem_gb": 16.07} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.10627452008575201, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 203}, "mem_gb": 15.92} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08060978428038458, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 239}, "mem_gb": 16.02} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05341216823590609, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 254}, "mem_gb": 15.93} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.057884399864574276, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 241}, "mem_gb": 16.02} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07263648351286538, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.3046875, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 213}, "mem_gb": 16.05} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0615926475533129, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 221}, "mem_gb": 16.1} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06679541949632888, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 196}, "mem_gb": 16.06} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05176316303300361, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 270}, "mem_gb": 15.86} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05212113441683663, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 216}, "mem_gb": 16.04} +[eval step 60] sample: 'To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers that lie on the same diagonal as the number \\(7\\). The spiral pattern starts at the center and' +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07816804139691716, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 200}, "mem_gb": 16.01} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05035967906449611, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 206}, "mem_gb": 15.96} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.044304009293485436, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.23828125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 234}, "mem_gb": 15.99} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05535436973415005, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 215}, "mem_gb": 16.0} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.044907201238783695, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 255}, "mem_gb": 15.99} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05459509837126049, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 250}, "mem_gb": 15.87} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.050126402091002095, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 242}, "mem_gb": 16.04} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07252040647125492, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 199}, "mem_gb": 16.05} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08199389099515975, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 208}, "mem_gb": 16.08} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06270051176787043, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 208}, "mem_gb": 16.02} +[eval step 70] sample: 'To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers that lie on the same diagonal as the number 7. The spiral pattern starts at the center and mov' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06326160759705429, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 209}, "mem_gb": 16.17} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.047964732579079766, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 235}, "mem_gb": 16.0} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05000873766591152, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 204}, "mem_gb": 15.99} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06726836371614288, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 208}, "mem_gb": 16.05} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.049516278548678384, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 240}, "mem_gb": 16.05} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05665072427910442, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.3046875, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 246}, "mem_gb": 16.04} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05669951267732928, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 243}, "mem_gb": 15.86} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07075669393978702, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 208}, "mem_gb": 16.06} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05974680195176043, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 219}, "mem_gb": 16.05} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0652965749655074, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 211}, "mem_gb": 16.06} +[eval step 80] sample: 'To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers that lie on the same diagonal as the number 7. The spiral pattern starts at the center and mov' +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05211935499627143, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 232}, "mem_gb": 16.02} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.059444162550428885, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 214}, "mem_gb": 16.05} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.055917505533155054, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.5, "frames": {"chat": 226}, "mem_gb": 15.94} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.055923757360627255, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 210}, "mem_gb": 16.06} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.047685957645904276, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.24609375, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 218}, "mem_gb": 15.88} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04715602353129846, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.2421875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 233}, "mem_gb": 16.03} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0596792153040568, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 215}, "mem_gb": 16.05} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05345027676153307, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 233}, "mem_gb": 16.04} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.050646483299819134, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 209}, "mem_gb": 15.99} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.044942502258066085, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 262}, "mem_gb": 15.92} +[eval step 90] sample: 'To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers that lie on the same diagonal as the number 7. The spiral pattern starts at the center and mov' +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.045185356042953206, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 249}, "mem_gb": 16.01} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06120699836833713, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 227}, "mem_gb": 16.04} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.053615750029784005, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 221}, "mem_gb": 16.04} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.048285325485731785, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 234}, "mem_gb": 16.06} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04676177461682819, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 213}, "mem_gb": 16.0} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.044436498762403305, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.23046875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 213}, "mem_gb": 15.94} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.052153425100895885, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 234}, "mem_gb": 15.97} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05320698270440723, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 222}, "mem_gb": 16.04} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.051590269053028895, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.2431640625, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.1, "frames": {"chat": 227}, "mem_gb": 16.05} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.057174564701706794, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 218}, "mem_gb": 16.09} +[eval step 100] sample: 'To solve this problem, we need to understand the structure of the spiral pattern and how the numbers are placed on the grid. The spiral pattern starts at the center and moves outward, forming a diamon' +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep50_s1225/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05653364848041286, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 223}, "mem_gb": 16.06} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05626137313898653, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.248046875, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 206}, "mem_gb": 16.05} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04629964479301125, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.24609375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 213}, "mem_gb": 15.97} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06836010164196292, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 223}, "mem_gb": 15.91} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04960931596377244, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.2333984375, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 227}, "mem_gb": 16.02} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04606072457487074, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.2431640625, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 253}, "mem_gb": 16.05} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.055348847734757387, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 216}, "mem_gb": 16.05} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03895854299830583, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 205}, "mem_gb": 16.02} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04872952806249571, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.2197265625, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 207}, "mem_gb": 16.11} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.044403939206223, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.20703125, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 215}, "mem_gb": 16.03} +[eval step 110] sample: 'To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers that lie on the same diagonal as the number 7. The spiral pattern starts at the center and mov' +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.035836265381332486, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.2275390625, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 228}, "mem_gb": 16.05} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03942134256685773, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.2109375, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 221}, "mem_gb": 16.09} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.029532018057101716, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.181640625, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 254}, "mem_gb": 15.89} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03353456746751132, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.248046875, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 210}, "mem_gb": 16.01} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.033196155026756845, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.203125, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 226}, "mem_gb": 15.97} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03837967902193001, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.2177734375, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 212}, "mem_gb": 16.04} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04279236014199753, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.2158203125, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 211}, "mem_gb": 15.97} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04495121304591497, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.2314453125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 196}, "mem_gb": 16.02} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03516720853671432, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.2021484375, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 212}, "mem_gb": 16.04} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03228002276973954, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.1904296875, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 244}, "mem_gb": 15.95} +[eval step 120] sample: 'To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers that lie on the same diagonal as the number 7. The spiral pattern starts at the center and mov' +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03490392869187829, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 222}, "mem_gb": 16.0} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03468469969631793, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.1943359375, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 218}, "mem_gb": 16.05} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03472803884683332, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.19921875, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 252}, "mem_gb": 15.92} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04160250526641806, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 202}, "mem_gb": 16.1} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04120717598129995, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 237}, "mem_gb": 16.05} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03709639240807543, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.203125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 234}, "mem_gb": 16.03} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03265363146445403, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 215}, "mem_gb": 16.05} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02957465929450312, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.181640625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 234}, "mem_gb": 15.98} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.031586363252205776, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.1806640625, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 216}, "mem_gb": 16.03} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03465553052170047, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.189453125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 210}, "mem_gb": 16.0} +[eval step 130] sample: 'To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers that lie on the same diagonal as the number 7. The spiral pattern starts at the center and mov' +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03964869703074607, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.2177734375, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 199}, "mem_gb": 16.05} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03579281948882466, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.2099609375, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 210}, "mem_gb": 16.06} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.033542947825191856, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 225}, "mem_gb": 16.0} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.032140976130838196, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.1826171875, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 253}, "mem_gb": 15.9} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03224871880656574, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.1796875, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 247}, "mem_gb": 16.02} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03395894770209367, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.1806640625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 238}, "mem_gb": 16.02} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03957084504778807, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.2353515625, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 235}, "mem_gb": 16.04} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04128963355836458, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.228515625, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.5, "frames": {"chat": 215}, "mem_gb": 16.01} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03190090727764182, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 268}, "mem_gb": 16.02} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.035086253207425276, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.201171875, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 228}, "mem_gb": 16.05} +[eval step 140] sample: 'To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers that lie on the same diagonal as the number 7. The spiral pattern starts at the center and mov' +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.033764375670044686, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.1982421875, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 252}, "mem_gb": 15.98} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03700443701595068, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.2119140625, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 223}, "mem_gb": 16.06} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04124379948658558, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.20703125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 226}, "mem_gb": 16.04} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04097800794257006, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 208}, "mem_gb": 16.09} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0317224430399326, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.1904296875, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 240}, "mem_gb": 15.98} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.037275905701781936, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.1826171875, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 222}, "mem_gb": 15.97} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03051341254238505, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.173828125, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 236}, "mem_gb": 16.04} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03466207155267087, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 217}, "mem_gb": 16.01} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03411953383701233, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.2314453125, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 252}, "mem_gb": 15.92} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.034074287171739465, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 222}, "mem_gb": 16.04} +[eval step 150] sample: 'To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers that lie on the same diagonal as the number 7. Then, we will determine which of these numbers ' +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep50_s1225/step0150 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▃▄▅▄▃▂▆▄▇▅▅▅▄▆▅▅▄▇▃▁▄█▇▃▇▇▆▅▆▄▅▂▇▅▂▁▅▄▆▅ +wandb: cumulative_loss_tokens ▁▁▁▁▁▁▂▂▂▂▂▂▂▂▂▃▃▃▃▃▃▃▄▄▄▄▄▄▄▅▅▅▅▅▅▆▆▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅██████████ +wandb: finish_rate ▁▅▆▂▅▃▂▃▅▆▅▇▂▂█▁▂▁▆▂▇▄▆▃▄▄▅▃▃▅▂▅▄▄▄▇▆█▄█ +wandb: forward_topk_kl █▄▄▃▃▂▂▂▂▂▂▂▂▂▁▁▁▁▂▁▂▁▁▁▁▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm ███▆▃▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂██████████████████████████████████████ +wandb: mem_gb ▄▆▆▂▆▆█▅▆▆▃▆▁▆▃▇▆▄▂▆▆▆▆▅▅▆▅▆▄▅▃▅▆▄▆▅▅▆▄▃ +wandb: step ▁▁▁▂▂▂▂▃▃▃▄▄▄▄▄▅▅▅▅▅▅▅▆▆▆▆▆▆▆▆▆▇▇▇▇▇████ +wandb: t_data_s ▁▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 540.5 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.847 +wandb: forward_topk_kl 0.03407 +wandb: grad_norm 0.2041 +wandb: lr 3e-05 +wandb: mem_gb 16.04 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run reap-math-keep50-s1225 at: https://wandb.ai/hbfreed/glean-grid/runs/3wq993z8 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/reap_keep50_s1225/wandb/run-20260716_014634-3wq993z8/logs +{ + "correct": 768, + "accuracy": 0.5822592873388931, + "finished": 1300, + "finish_rate": 0.9855951478392722, + "mean_completion_tokens": 123.63381349507202 +} +saved item-level results -> outputs/evals/grid_math/reap_keep50_s1225_step100_chat.json +{ + "correct": 769, + "accuracy": 0.5830174374526156, + "finished": 1306, + "finish_rate": 0.9901440485216073, + "mean_completion_tokens": 124.30326004548901 +} +saved item-level results -> outputs/evals/grid_math/reap_keep50_s1225_step150_chat.json diff --git a/healed/grid_math/reap_keep50_s1226.console.log b/healed/grid_math/reap_keep50_s1226.console.log new file mode 100644 index 0000000000000000000000000000000000000000..d8017eaaa7ceb0a25ed3edfa58ff519a280a5045 --- /dev/null +++ b/healed/grid_math/reap_keep50_s1226.console.log @@ -0,0 +1,231 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/reap_keep50_s1226/wandb/run-20260716_014602-qt9aq0ed +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run reap-math-keep50-s1226 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/qt9aq0ed + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/grid_math/reap_keep50_s1226/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.10459128459574034, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 223}, "mem_gb": 15.91} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08826250612973235, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 227}, "mem_gb": 16.02} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07563404675796628, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 253}, "mem_gb": 16.05} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06886030516719135, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 216}, "mem_gb": 16.05} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06066073822774924, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 205}, "mem_gb": 16.02} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07448192878928966, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 207}, "mem_gb": 16.11} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07085500255872806, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 215}, "mem_gb": 16.03} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.054219987386542684, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 228}, "mem_gb": 16.05} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06803286743182689, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 221}, "mem_gb": 16.09} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04641861871878306, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.2392578125, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 254}, "mem_gb": 15.89} +[eval step 60] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.\n -" +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05023701873199704, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 210}, "mem_gb": 16.01} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.051724348243310424, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 226}, "mem_gb": 15.97} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06272758410156239, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 212}, "mem_gb": 16.04} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06369718516276529, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 211}, "mem_gb": 15.97} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0647698606271452, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 196}, "mem_gb": 16.02} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.059465903049598756, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 212}, "mem_gb": 16.04} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0527960841421814, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 244}, "mem_gb": 15.95} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05209585945080034, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.2314453125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 222}, "mem_gb": 16.0} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05374352757440259, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 218}, "mem_gb": 16.05} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05794376976617301, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 252}, "mem_gb": 15.92} +[eval step 70] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.\n -" +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06893676832563554, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 202}, "mem_gb": 16.1} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07893477528166647, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 237}, "mem_gb": 16.05} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06014306015539914, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 234}, "mem_gb": 16.03} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.047996336566330865, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 215}, "mem_gb": 16.05} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.044697797822409, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.2470703125, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 234}, "mem_gb": 15.98} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04588704365013788, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 216}, "mem_gb": 16.03} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05109065430917156, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.2451171875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 210}, "mem_gb": 16.0} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05233408396915765, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 199}, "mem_gb": 16.05} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04801637768183525, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.23828125, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 210}, "mem_gb": 16.06} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.053485469416653116, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 225}, "mem_gb": 16.0} +[eval step 80] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.\n -" +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05053567462177016, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 253}, "mem_gb": 15.9} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04922324699338836, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.2373046875, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 247}, "mem_gb": 16.02} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.050183612028866384, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.2353515625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 238}, "mem_gb": 16.02} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.054622684019478035, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 235}, "mem_gb": 16.04} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05814677753490396, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 215}, "mem_gb": 16.01} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05711578755577405, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 268}, "mem_gb": 16.02} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.051577915780417, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 228}, "mem_gb": 16.05} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.055417160641759014, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 252}, "mem_gb": 15.98} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05028329864336799, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 223}, "mem_gb": 16.06} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07661831421251408, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 226}, "mem_gb": 16.04} +[eval step 90] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the midpoints of its sides.\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.' +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.061045037919143216, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 208}, "mem_gb": 16.09} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04929783342856293, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 240}, "mem_gb": 15.98} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05484731227552208, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.2470703125, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 222}, "mem_gb": 15.97} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04571815300624973, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.2333984375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 236}, "mem_gb": 16.04} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04882919267296481, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 217}, "mem_gb": 16.01} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05304940091329627, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 252}, "mem_gb": 15.92} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0453409172443673, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.2431640625, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 222}, "mem_gb": 16.04} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04914017798041459, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.901, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 242}, "mem_gb": 15.92} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.06614599686032162, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 219}, "mem_gb": 15.98} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04687103871277844, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.2412109375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 223}, "mem_gb": 15.99} +[eval step 100] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.\n -" +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep50_s1226/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05508217736965356, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 232}, "mem_gb": 15.99} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05771918959524482, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.832, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 220}, "mem_gb": 16.05} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.05670769996464563, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 210}, "mem_gb": 16.09} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04982835436208795, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.2431640625, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 226}, "mem_gb": 16.02} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.049105689908145, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.248046875, "lr": 3e-05, "finish_rate": 0.741, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 212}, "mem_gb": 16.04} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04600744808347275, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.2470703125, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 236}, "mem_gb": 16.06} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.033932253168631965, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.928, "comp_len": 454.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 264}, "mem_gb": 15.93} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0591216995248571, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 229}, "mem_gb": 16.03} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.033564152960277475, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.197265625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 465.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 258}, "mem_gb": 15.9} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05125372361148087, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 208}, "mem_gb": 16.06} +[eval step 110] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.\n -" +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03912771535598052, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.2197265625, "lr": 3e-05, "finish_rate": 0.88, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 249}, "mem_gb": 15.97} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03329837650329185, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 220}, "mem_gb": 16.04} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03639425171962939, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.216796875, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 223}, "mem_gb": 16.03} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03565040662499765, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.2001953125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 221}, "mem_gb": 16.04} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03295701659352829, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 230}, "mem_gb": 15.95} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04046579788605062, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 214}, "mem_gb": 16.02} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0523497827064246, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.244140625, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 214}, "mem_gb": 16.04} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04217318628588691, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.2109375, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 210}, "mem_gb": 16.08} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04559627010982173, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 214}, "mem_gb": 16.04} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.039164316506125035, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.791, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 215}, "mem_gb": 16.0} +[eval step 120] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.\n -" +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04476614243153017, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.2177734375, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 208}, "mem_gb": 16.04} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03574293609918095, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.789, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 218}, "mem_gb": 15.92} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03212748550867352, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.177734375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 233}, "mem_gb": 15.94} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.031179524623261144, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 231}, "mem_gb": 15.98} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.037280551470660915, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.2109375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 235}, "mem_gb": 16.18} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03532052122700649, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.181640625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 216}, "mem_gb": 16.03} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03285237047360279, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.1943359375, "lr": 3e-05, "finish_rate": 0.831, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 237}, "mem_gb": 16.06} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0456035717476625, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.2177734375, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 203}, "mem_gb": 16.06} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03478093251015525, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.873, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 236}, "mem_gb": 16.08} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03525485854660316, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.189453125, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 214}, "mem_gb": 16.05} +[eval step 130] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.\n -" +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.039782691275918235, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.2060546875, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 209}, "mem_gb": 16.04} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03294830878264426, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 256}, "mem_gb": 16.05} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03306894852546975, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 230}, "mem_gb": 15.91} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03653117787890757, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.20703125, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 230}, "mem_gb": 16.1} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04157050093587798, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 227}, "mem_gb": 16.0} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04171398106655106, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.2119140625, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 208}, "mem_gb": 16.06} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.04136487604933791, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.203125, "lr": 3e-05, "finish_rate": 0.699, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 206}, "mem_gb": 16.08} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0372208833127593, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.1982421875, "lr": 3e-05, "finish_rate": 0.82, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 228}, "mem_gb": 15.95} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03776092570159817, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.1904296875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 224}, "mem_gb": 16.04} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.037290410421757646, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.66, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 200}, "mem_gb": 16.08} +[eval step 140] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.\n -" +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03674508915361172, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.1982421875, "lr": 3e-05, "finish_rate": 0.714, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 196}, "mem_gb": 16.06} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03390032124063, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.1865234375, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 223}, "mem_gb": 16.04} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03720896863811649, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.21875, "lr": 3e-05, "finish_rate": 0.869, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 213}, "mem_gb": 15.93} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03197799935330792, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.1787109375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 232}, "mem_gb": 15.98} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03063049944328765, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.173828125, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 223}, "mem_gb": 15.97} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03251715304492973, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.1787109375, "lr": 3e-05, "finish_rate": 0.85, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 233}, "mem_gb": 16.07} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.038545895068844156, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.1923828125, "lr": 3e-05, "finish_rate": 0.816, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 217}, "mem_gb": 16.06} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05470696568132068, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.240234375, "lr": 3e-05, "finish_rate": 0.752, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 202}, "mem_gb": 16.12} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03262044265196504, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.1884765625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 253}, "mem_gb": 15.98} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03266540290627939, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.1953125, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 231}, "mem_gb": 15.98} +[eval step 150] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.\n -" +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep50_s1226/step0150 +wandb: updating run metadata +wandb: uploading summary, console lines 171-171 +wandb: +wandb: Run history: +wandb: comp_len ▅▂▆▂▂▃▅▆▅▄▁▅▇▆▅▅▆█▆▂▅▆▄▃▁▅▃▅▁▄▆▂▄▅▅▁▄▆▄▃ +wandb: cumulative_loss_tokens ▁▁▁▂▂▂▂▂▂▂▂▂▃▃▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▆▆▇█ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅███████████████ +wandb: finish_rate ▅▇▄▂▂▅▆▄▅▆▇▇▅▅▄▃▅▅▇▄▂▅▄▂██▇▄▇▄▂▇▄▄▇▇▁▅▆▇ +wandb: forward_topk_kl █▅▃▃▃▂▂▂▂▂▂▂▂▂▂▁▂▁▂▁▁▁▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm █▃▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁███████████████████████████████████████ +wandb: mem_gb ▁▄▆█▅▆▂▅▅▅▆▃▄▃▅▅▇▂▅▄▅▆▅▅▄▅▃▄▆▃▃▄▅▅▃▄▄▆▆▄ +wandb: step ▁▁▂▂▂▃▃▃▃▃▃▃▃▃▃▃▄▄▄▄▄▄▄▄▄▅▅▅▆▆▆▇▇▇▇▇████ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 519.5 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.879 +wandb: forward_topk_kl 0.03267 +wandb: grad_norm 0.19531 +wandb: lr 3e-05 +wandb: mem_gb 15.98 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run reap-math-keep50-s1226 at: https://wandb.ai/hbfreed/glean-grid/runs/qt9aq0ed +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/reap_keep50_s1226/wandb/run-20260716_014602-qt9aq0ed/logs +{ + "correct": 776, + "accuracy": 0.5883244882486732, + "finished": 1307, + "finish_rate": 0.9909021986353298, + "mean_completion_tokens": 123.34874905231236 +} +saved item-level results -> outputs/evals/grid_math/reap_keep50_s1226_step100_chat.json +{ + "correct": 780, + "accuracy": 0.5913570887035633, + "finished": 1304, + "finish_rate": 0.9886277482941622, + "mean_completion_tokens": 126.97573919636088 +} +saved item-level results -> outputs/evals/grid_math/reap_keep50_s1226_step150_chat.json diff --git a/healed/grid_math/reap_keep75_s1224.console.log b/healed/grid_math/reap_keep75_s1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..65332e8365912727478b514d107f70d066cb7af0 --- /dev/null +++ b/healed/grid_math/reap_keep75_s1224.console.log @@ -0,0 +1,324 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run clpi70s7 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/reap_keep75_s1224/wandb/run-20260716_171845-clpi70s7 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run reap-math-keep75-s1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/clpi70s7 + Loading checkpoint shards: 0%| | 0/3 [00:00 outputs/healed/grid_math/reap_keep75_s1224/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02873803478294673, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 222}, "mem_gb": 22.05} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.029475814702517044, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 455.7, "frames": {"chat": 235}, "mem_gb": 22.1} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03140903974018681, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 456.1, "frames": {"chat": 208}, "mem_gb": 22.06} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01991006130973498, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 452.3, "frames": {"chat": 191}, "mem_gb": 22.1} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017155938783214274, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 445.7, "frames": {"chat": 219}, "mem_gb": 22.09} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.022398129164334386, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.248046875, "lr": 3e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 480.8, "frames": {"chat": 207}, "mem_gb": 22.1} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03531387661455665, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 446.3, "frames": {"chat": 208}, "mem_gb": 22.05} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019110731818852946, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 441.8, "frames": {"chat": 219}, "mem_gb": 22.09} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.020053759875210624, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.244140625, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 468.8, "frames": {"chat": 246}, "mem_gb": 21.97} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0288994662804141, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 460.8, "frames": {"chat": 206}, "mem_gb": 22.12} +[eval step 60] sample: "To solve the given system of equations, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(r\\) such that each letter represents a non-zero digit. Let's break down the problem step-by-step" +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.021922170959234547, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.2353515625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 427.6, "frames": {"chat": 233}, "mem_gb": 22.1} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02431659371315812, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 436.1, "frames": {"chat": 229}, "mem_gb": 21.96} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018058950587594883, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.2197265625, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 449.1, "frames": {"chat": 236}, "mem_gb": 22.0} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.021461314609615752, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.2197265625, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 433.3, "frames": {"chat": 239}, "mem_gb": 21.88} +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run s8ctve75 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/reap_keep75_s1224/wandb/run-20260716_194622-s8ctve75 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run reap-math-keep75-s1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/s8ctve75 + Loading checkpoint shards: 0%| | 0/3 [00:00 outputs/healed/grid_math/reap_keep75_s1224/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018019333380716852, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 232}, "mem_gb": 21.98} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019176917587368128, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.2197265625, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 210}, "mem_gb": 22.03} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018526647040364334, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.2197265625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 217}, "mem_gb": 22.0} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017947978142862364, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.19921875, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 224}, "mem_gb": 22.12} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.021901778530251857, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.23828125, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 203}, "mem_gb": 21.97} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017482746841735206, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 239}, "mem_gb": 22.07} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010781270754368355, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.142578125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 254}, "mem_gb": 21.98} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012419780704270427, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.171875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 241}, "mem_gb": 22.07} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015673370206107696, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.166015625, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 213}, "mem_gb": 22.1} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014789011545067964, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.2119140625, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 221}, "mem_gb": 22.15} +[eval step 110] sample: "To solve the given system of equations, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(r\\) such that each letter represents a non-zero digit. Let's break down the problem step-by-step" +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015781028156393827, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.1943359375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 196}, "mem_gb": 22.11} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010667995535299027, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.1279296875, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 270}, "mem_gb": 21.91} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012989928167418112, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.1572265625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 216}, "mem_gb": 22.09} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013924577494691281, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.1513671875, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 200}, "mem_gb": 22.06} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013322410390495013, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.1572265625, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 206}, "mem_gb": 22.01} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011119994657541004, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.1396484375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 234}, "mem_gb": 22.04} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012775608483022855, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.1416015625, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 215}, "mem_gb": 22.05} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013810691532116228, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.173828125, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 255}, "mem_gb": 22.04} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012154009584007629, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.1611328125, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 250}, "mem_gb": 21.92} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011215298393421108, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.140625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 242}, "mem_gb": 22.09} +[eval step 120] sample: "To solve the given system of equations, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(r\\) such that each letter represents a non-zero digit. Let's break down the problem step-by-step" +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01326268654340723, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.14453125, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 199}, "mem_gb": 22.1} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.018925391417974606, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.1865234375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 208}, "mem_gb": 22.13} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01369877224289424, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.2001953125, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 208}, "mem_gb": 22.07} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01566584520537872, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 209}, "mem_gb": 22.22} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011354163011299292, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.1435546875, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 235}, "mem_gb": 22.05} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013031885967287235, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.1728515625, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 204}, "mem_gb": 22.04} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.016365782146974622, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.20703125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 208}, "mem_gb": 22.1} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012403703387636536, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.158203125, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 240}, "mem_gb": 22.1} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01273732632562751, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.1630859375, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 246}, "mem_gb": 22.09} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012520373641039865, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.1396484375, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 243}, "mem_gb": 21.91} +[eval step 130] sample: "To solve the given system of equations, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(r\\) such that each letter represents a non-zero digit. Let's break down the problem step-by-step" +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01572654613229679, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.15625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 208}, "mem_gb": 22.11} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.017281583469605538, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.171875, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.9, "frames": {"chat": 219}, "mem_gb": 22.1} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015536821740671681, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.173828125, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 211}, "mem_gb": 22.11} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.016642906039534135, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.2080078125, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 232}, "mem_gb": 22.07} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.019927484658709728, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 214}, "mem_gb": 22.1} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012803300328789433, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.146484375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 226}, "mem_gb": 21.99} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012066151755136282, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.1474609375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 210}, "mem_gb": 22.11} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012860532379080542, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.1591796875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 218}, "mem_gb": 21.93} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012094843646103982, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.1396484375, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 233}, "mem_gb": 22.08} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015558591982846459, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.169921875, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 215}, "mem_gb": 22.1} +[eval step 140] sample: "To solve the given system of equations, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(r\\) such that each letter represents a non-zero digit. Let's break down the problem step-by-step" +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012454204010121369, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.146484375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 233}, "mem_gb": 22.09} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013477057802421042, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 209}, "mem_gb": 22.04} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009968187428083426, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.123046875, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.8, "frames": {"chat": 262}, "mem_gb": 21.97} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013083611388411373, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.15234375, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 249}, "mem_gb": 22.06} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014526255074034756, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.1748046875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.5, "frames": {"chat": 227}, "mem_gb": 22.09} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011909497406838152, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.14453125, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 221}, "mem_gb": 22.09} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012006994549774875, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.1533203125, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 234}, "mem_gb": 22.11} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012652689880512965, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.177734375, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 213}, "mem_gb": 22.05} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011978251870415018, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.1513671875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 213}, "mem_gb": 21.99} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01043063024947187, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.13671875, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 234}, "mem_gb": 22.02} +[eval step 150] sample: "To solve the given system of equations, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(r\\) such that each letter represents a non-zero digit. Let's break down the problem step-by-step" +checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep75_s1224/step0150 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▅▇▇▇▃▄▃▃▅▂▇▅▄▇▅▆▇▆▂▆▃▁▄▆▆▂▆▅█▇▂▃█▇▃▄▄▆▄▄ +wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▂▂▂▃▃▃▄▄▄▄▄▄▅▅▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇██ +wandb: epoch ▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅███████████████ +wandb: finish_rate ▄▁▅▄▆▆▇█▂▄▁▂▅▅▄█▃█▅▆▆▄█▆▃▅▂▇▇██▇▁▂▂▇▄▃▇▇ +wandb: forward_topk_kl █▃▅▅▅▄▄▃▆▅▆▅▄▃▄▄▃▄▃▃▃▄▄▅▃▃▁▂▂▁▄▂▃▂▂▃▄▃▃▁ +wandb: grad_norm █▆▆▅▅▄▆▇▆▆▅▄▄▄▆▅▆▃▃▇▄▄▃▂▅▂▁▂▁▁▃▄▂▂▃▁▂▁▁▂ +wandb: lr ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: mem_gb ▂▆▆▆▄▆▃▂▄▇▆█▆▅▆▇▆▄▆▇▅▆▁▄▄▆▅▅▆▆█▆▆▆▆▆▃▆▄▆ +wandb: step ▁▁▁▁▁▂▂▃▃▃▃▃▄▄▄▅▅▅▅▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇▇████ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 512.8 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.906 +wandb: forward_topk_kl 0.01043 +wandb: grad_norm 0.13672 +wandb: lr 3e-05 +wandb: mem_gb 22.02 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run reap-math-keep75-s1224 at: https://wandb.ai/hbfreed/glean-grid/runs/f1ktljdw +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/reap_keep75_s1224/wandb/run-20260716_210030-f1ktljdw/logs +{ + "correct": 894, + "accuracy": 0.6777862016679302, + "finished": 1315, + "finish_rate": 0.9969673995451099, + "mean_completion_tokens": 112.61410159211523 +} +saved item-level results -> outputs/evals/grid_math/reap_keep75_s1224_step100_chat.json +{ + "correct": 903, + "accuracy": 0.6846095526914329, + "finished": 1316, + "finish_rate": 0.9977255496588324, + "mean_completion_tokens": 111.83548142532221 +} +saved item-level results -> outputs/evals/grid_math/reap_keep75_s1224_step150_chat.json diff --git a/healed/grid_math/uniform_keep25_s1224.console.log b/healed/grid_math/uniform_keep25_s1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..fb768686ca2e6a36aaf8eef68eda7e88df08714a --- /dev/null +++ b/healed/grid_math/uniform_keep25_s1224.console.log @@ -0,0 +1,231 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run rm3hkgkc +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/uniform_keep25_s1224/wandb/run-20260716_054248-rm3hkgkc +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run uniform-math-keep25-s1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/rm3hkgkc +12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 2.09B | teacher overlap=False +{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2459722065463663, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 14.625, "lr": 6e-06, "finish_rate": 0.907, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.6, "frames": {"chat": 236}, "mem_gb": 9.78} +The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results. +[eval step 1] sample: 'The value of $p$ is \\(\\box{5}\\).' +{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.3142343253110846, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 15.375, "lr": 9e-06, "finish_rate": 0.781, "comp_len": 558.1, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 27.7, "frames": {"chat": 215}, "mem_gb": 10.01} +{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.3116438292915622, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 13.9375, "lr": 1.2e-05, "finish_rate": 0.825, "comp_len": 553.0, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 217}, "mem_gb": 9.88} +{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.1271550920541087, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 11.1875, "lr": 1.5e-05, "finish_rate": 0.8, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 205}, "mem_gb": 9.94} +{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8921521788805723, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 6.40625, "lr": 1.8e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 229}, "mem_gb": 9.92} +{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8736880722011129, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 4.8125, "lr": 2.1e-05, "finish_rate": 0.812, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 223}, "mem_gb": 9.98} +{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7521487558588386, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 3.453125, "lr": 2.4e-05, "finish_rate": 0.708, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 202}, "mem_gb": 10.02} +{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7010229941956699, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 3.1875, "lr": 2.7000000000000002e-05, "finish_rate": 0.77, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 209}, "mem_gb": 9.99} +{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6002694400727749, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 2.453125, "lr": 3e-05, "finish_rate": 0.885, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 227}, "mem_gb": 9.97} +{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5602013669068615, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 1.8203125, "lr": 3e-05, "finish_rate": 0.848, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 230}, "mem_gb": 10.05} +[eval step 10] sample: "To solve the problem, we need to determine the value of \\(p\\) given the following equations:\n\n\\[\na + b = k \\\\\nk + m = p \\\\\np + a = r \\\\\nb + m + r = 18\n\\]\n\nLet's break down the problem step-" +{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5372615832733612, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 1.546875, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 231}, "mem_gb": 9.89} +{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5131440531161924, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 1.328125, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.8, "frames": {"chat": 245}, "mem_gb": 9.97} +{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.47872232480570676, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 1.234375, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 210}, "mem_gb": 9.98} +{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.537974064292262, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 1.2578125, "lr": 3e-05, "finish_rate": 0.758, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.0, "frames": {"chat": 211}, "mem_gb": 9.98} +{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4437455804655949, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.9296875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 221}, "mem_gb": 10.03} +{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4362137271172057, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.984375, "lr": 3e-05, "finish_rate": 0.912, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 250}, "mem_gb": 9.84} +{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4411819472976029, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.8984375, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.6, "frames": {"chat": 229}, "mem_gb": 10.02} +{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.37396339893067876, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.83984375, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 250}, "mem_gb": 9.99} +{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.40561543378531933, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.79296875, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 231}, "mem_gb": 9.87} +{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4002509487184385, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.75390625, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 224}, "mem_gb": 9.91} +[eval step 20] sample: "To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(r\\). Let's break down the problem step-by-step:\n\n1. **Define Variables:**\n - Let \\(a\\) be the first digit" +{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.36296452349474034, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.80078125, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 212}, "mem_gb": 9.95} +{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.34317847680710256, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.75390625, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 238}, "mem_gb": 9.91} +{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.364331378469492, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.81640625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.8, "frames": {"chat": 257}, "mem_gb": 9.78} +{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3275189952706297, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 227}, "mem_gb": 9.97} +{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3667670934761564, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 228}, "mem_gb": 10.0} +{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3577250474138806, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32106263146164515, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.41402777843773364, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.734375, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 197}, "mem_gb": 10.08} +{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3724903223272413, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.68359375, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 239}, "mem_gb": 9.84} +{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.35697053606547413, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.6875, "lr": 3e-05, "finish_rate": 0.83, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 224}, "mem_gb": 9.88} +[eval step 30] sample: 'To solve the problem, we need to determine the value of \\( p \\) given the equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}' +{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.31142014599790174, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 217}, "mem_gb": 10.0} +{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.29815615977446236, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 241}, "mem_gb": 10.0} +{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3220009417420874, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 218}, "mem_gb": 9.97} +{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.31028348484759527, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 206}, "mem_gb": 9.98} +{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3256828921566407, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.69140625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.3, "frames": {"chat": 232}, "mem_gb": 10.03} +{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3209710501977553, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.6171875, "lr": 3e-05, "finish_rate": 0.771, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.0, "frames": {"chat": 218}, "mem_gb": 10.05} +{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3221005113951862, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.779, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.0, "frames": {"chat": 213}, "mem_gb": 10.01} +{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32981064673662186, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.59765625, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 248}, "mem_gb": 9.98} +{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.30937803875269987, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.6171875, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 218}, "mem_gb": 10.04} +{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2680865074227254, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.54296875, "lr": 3e-05, "finish_rate": 0.851, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 221}, "mem_gb": 9.99} +[eval step 40] sample: 'To solve the system of equations given by:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe need to determine the value of' +{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.29091000881840784, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.5546875, "lr": 3e-05, "finish_rate": 0.894, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 236}, "mem_gb": 9.92} +{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.29280417794175445, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.5625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 246}, "mem_gb": 9.84} +{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2934912766356021, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 234}, "mem_gb": 10.1} +{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23609974780020615, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.748, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.0, "frames": {"chat": 202}, "mem_gb": 9.98} +{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.26752842073862754, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.7, "frames": {"chat": 217}, "mem_gb": 9.99} +{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28959610045862694, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.866, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 224}, "mem_gb": 9.99} +{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3118727909699082, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.753, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 215}, "mem_gb": 10.01} +{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.26413657319024203, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.5625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.7, "frames": {"chat": 259}, "mem_gb": 9.93} +{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.29531795247793197, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.55078125, "lr": 3e-05, "finish_rate": 0.829, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 210}, "mem_gb": 9.99} +{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32368664257427054, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.6171875, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.1, "frames": {"chat": 213}, "mem_gb": 10.05} +[eval step 50] sample: 'To solve the system of equations given by:\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\nwe need to determine the value of \\(p' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep25_s1224/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28201613237416995, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.546875, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.7, "frames": {"chat": 222}, "mem_gb": 9.95} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24275245352555067, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 235}, "mem_gb": 10.01} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27474053088929506, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.6640625, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 208}, "mem_gb": 9.96} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.27640588794089854, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 1.671875, "lr": 3e-05, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 191}, "mem_gb": 10.0} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2337344744838774, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19981578330968816, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 207}, "mem_gb": 10.0} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2742745844999949, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 208}, "mem_gb": 9.96} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2056781067028021, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22058189589908966, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.6, "frames": {"chat": 246}, "mem_gb": 9.87} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2643624147929872, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.5625, "lr": 3e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 206}, "mem_gb": 10.02} +[eval step 60] sample: 'To solve the system of equations given by:\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\nwe need to find the value of \\(p' +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21315796338009338, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.546875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.8, "frames": {"chat": 233}, "mem_gb": 10.01} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23010029078846175, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.52734375, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 229}, "mem_gb": 9.87} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18971112331363063, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 236}, "mem_gb": 9.9} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22939992027953268, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 239}, "mem_gb": 9.79} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19626577739318213, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 241}, "mem_gb": 9.91} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2318629670790086, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 226}, "mem_gb": 9.87} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21305266705161582, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.478515625, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.7, "frames": {"chat": 234}, "mem_gb": 10.0} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2066091762378812, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 257}, "mem_gb": 9.99} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2726578070071836, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.2, "frames": {"chat": 208}, "mem_gb": 10.05} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22639581966890643, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.6, "frames": {"chat": 211}, "mem_gb": 10.02} +[eval step 70] sample: 'To solve the system of equations given by:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe need to determine the value of' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2497449489728858, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 227}, "mem_gb": 10.01} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23184565024909873, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.478515625, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.1, "frames": {"chat": 211}, "mem_gb": 9.98} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19929239737944057, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 238}, "mem_gb": 10.0} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21444108126734693, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 237}, "mem_gb": 10.04} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.228507239554512, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 208}, "mem_gb": 10.04} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22201900441398223, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 221}, "mem_gb": 10.12} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2176680883942172, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 232}, "mem_gb": 9.96} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20230479391242068, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.4, "frames": {"chat": 208}, "mem_gb": 10.0} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20621175042502582, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 227}, "mem_gb": 9.91} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23712326520072918, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 221}, "mem_gb": 9.94} +[eval step 80] sample: 'To solve the system of linear equations given by:\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\nwe need to determine the value of \\(' +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18516172153086713, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 232}, "mem_gb": 10.01} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19767518214893837, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 219}, "mem_gb": 10.01} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20814777844653776, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 195}, "mem_gb": 10.1} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21465807686001062, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 216}, "mem_gb": 10.0} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19052286817779143, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.796875, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.6, "frames": {"chat": 208}, "mem_gb": 9.89} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20568204772435128, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 235}, "mem_gb": 9.88} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2174682028145219, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 225}, "mem_gb": 9.99} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.25289742040882507, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.2, "frames": {"chat": 213}, "mem_gb": 10.08} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18773433045589674, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 257}, "mem_gb": 9.76} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21751305311309796, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.55078125, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 212}, "mem_gb": 10.03} +[eval step 90] sample: "To solve this system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) that satisfy all the given equations. Let's break down the problem step-by-step:\n\n1. " +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19270047811983773, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.6, "frames": {"chat": 221}, "mem_gb": 10.0} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19187464828671266, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 242}, "mem_gb": 10.0} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16515993953794242, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 220}, "mem_gb": 9.96} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19365566024774064, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 240}, "mem_gb": 9.86} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19843267694873115, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 206}, "mem_gb": 9.99} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1955687409762914, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 226}, "mem_gb": 10.0} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23913774186559023, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.55078125, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 244}, "mem_gb": 9.79} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24111862684388954, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 224}, "mem_gb": 10.01} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20014771827943623, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 271}, "mem_gb": 9.73} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18671405559678872, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 236}, "mem_gb": 10.01} +[eval step 100] sample: 'To solve the system of equations given by:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe need to find the values of' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep25_s1224/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20005840366501362, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 232}, "mem_gb": 9.88} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17052629617406057, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 210}, "mem_gb": 9.94} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16335184906447928, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 217}, "mem_gb": 9.91} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21374459005383153, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 224}, "mem_gb": 10.03} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24060933916370075, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.54296875, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 203}, "mem_gb": 9.87} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2103607431170841, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 239}, "mem_gb": 9.97} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16764343557308117, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.53125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 254}, "mem_gb": 9.88} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14560104954248915, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 241}, "mem_gb": 9.98} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21455759129527335, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 213}, "mem_gb": 10.01} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17589531892972687, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 221}, "mem_gb": 10.05} +[eval step 110] sample: 'To solve the system of linear equations given by:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe need to determine the values' +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18606396657917648, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 196}, "mem_gb": 10.01} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1608481796350951, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.8, "frames": {"chat": 270}, "mem_gb": 9.82} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14370967580489813, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 216}, "mem_gb": 10.0} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18907757144179196, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 200}, "mem_gb": 9.96} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1409036660529673, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 206}, "mem_gb": 9.91} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15345207378479342, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 234}, "mem_gb": 9.95} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1820456429388995, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 215}, "mem_gb": 9.96} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1530056454622497, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 255}, "mem_gb": 9.94} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16267353230559578, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 250}, "mem_gb": 9.82} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16486497175091255, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 242}, "mem_gb": 10.0} +[eval step 120] sample: 'To solve the system of equations given by:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &= 18\n\\end{align*}\n\\]\n\nwe need to determine the values of' +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19459200162775814, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20826055347186823, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 208}, "mem_gb": 10.04} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1550956260437922, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 208}, "mem_gb": 9.97} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17796036245978128, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.5, "frames": {"chat": 209}, "mem_gb": 10.12} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14426678808629512, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 235}, "mem_gb": 9.96} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1457931994455556, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 204}, "mem_gb": 9.95} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19706413115250568, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.48828125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1500166365404303, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.1, "frames": {"chat": 240}, "mem_gb": 10.0} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15483832397218794, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 246}, "mem_gb": 9.99} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16828932282235473, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 243}, "mem_gb": 9.82} +[eval step 130] sample: 'To solve the given system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) such that:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p' +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18825837670378387, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16242987946247062, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1970008016841486, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 211}, "mem_gb": 10.01} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14961124420731017, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.439453125, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 232}, "mem_gb": 9.98} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17444652401488275, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 214}, "mem_gb": 10.01} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15362181645246845, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.6, "frames": {"chat": 226}, "mem_gb": 9.9} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15746455868557097, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 1.0859375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.3, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.131130507747146, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 218}, "mem_gb": 9.83} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13699651974520335, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 2.265625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.8, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18451387127457808, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 215}, "mem_gb": 10.01} +[eval step 140] sample: 'To solve this problem, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) given the equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\n' +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.171622509614937, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1397777112826084, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 209}, "mem_gb": 9.94} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14401639284497747, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.4, "frames": {"chat": 262}, "mem_gb": 9.88} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14294547926882903, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 249}, "mem_gb": 9.96} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1826381297783926, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.3, "frames": {"chat": 227}, "mem_gb": 10.0} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13961569585020964, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.3, "frames": {"chat": 221}, "mem_gb": 10.0} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15722946502156557, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.2, "frames": {"chat": 234}, "mem_gb": 10.01} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12678958315073202, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.8515625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 213}, "mem_gb": 9.96} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1364368616912514, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 213}, "mem_gb": 9.9} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15012081333752722, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 1.0390625, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.7, "frames": {"chat": 234}, "mem_gb": 9.93} +[eval step 150] sample: "To solve this problem, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) such that the given equations hold true. Let's break down the problem step-by-step:\n\n1. **Understand t" +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep25_s1224/step0150 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▆▄▄▂▃▅▃▅▆▅▇▆▆▆▆▅▇▄▄▆▅▇▆▇▅▃█▁▆▂▇▇▇▅▆▅▆▄▆▆ +wandb: cumulative_loss_tokens ▁▁▁▂▂▂▂▂▂▂▃▃▃▃▃▃▃▃▄▄▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇█████ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅███████████ +wandb: finish_rate ▁▅▄▄▇▆▆▅▅▃▇▅▅▃▄█▃▃▄▁▄▁█▃█▆▅▂▆▅▆█▄▃▂▃▆▅▇▄ +wandb: forward_topk_kl █▆▅▄▃▃▃▃▃▂▂▂▂▂▂▁▂▂▂▂▂▁▂▁▂▁▂▂▂▂▁▁▁▂▁▁▁▁▁▁ +wandb: grad_norm █▂▂▂▂▂▁▁▁▁▁▁▁▁▁▁▁▁▃▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▄▂ +wandb: lr ▁▄▇█████████████████████████████████████ +wandb: mem_gb ▆▅▆▃▄▄▁▅▆▆▅▇█▆▆▇▆▅▆▆▃▇▆▆▆▆▆▆▆▃▆▅▅▆▇▂▆▆▂▆ +wandb: step ▁▁▂▂▂▂▃▃▃▃▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇████ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 512.8 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.906 +wandb: forward_topk_kl 0.15012 +wandb: grad_norm 1.03906 +wandb: lr 3e-05 +wandb: mem_gb 9.93 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run uniform-math-keep25-s1224 at: https://wandb.ai/hbfreed/glean-grid/runs/rm3hkgkc +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/uniform_keep25_s1224/wandb/run-20260716_054248-rm3hkgkc/logs +{ + "correct": 279, + "accuracy": 0.21152388172858225, + "finished": 1256, + "finish_rate": 0.9522365428354814, + "mean_completion_tokens": 158.7035633055345 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep25_s1224_step100_chat.json +{ + "correct": 277, + "accuracy": 0.2100075815011372, + "finished": 1271, + "finish_rate": 0.9636087945413192, + "mean_completion_tokens": 150.8013646702047 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep25_s1224_step150_chat.json diff --git a/healed/grid_math/uniform_keep25_s1225.console.log b/healed/grid_math/uniform_keep25_s1225.console.log new file mode 100644 index 0000000000000000000000000000000000000000..7d45522d5d914f116b311117fe6c62cc2e1ae058 --- /dev/null +++ b/healed/grid_math/uniform_keep25_s1225.console.log @@ -0,0 +1,230 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/uniform_keep25_s1225/wandb/run-20260716_051726-a8l8mpfr +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run uniform-math-keep25-s1225 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/a8l8mpfr +12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 2.09B | teacher overlap=False +{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.319479045200348, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 14.9375, "lr": 6e-06, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.2, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 191}, "mem_gb": 9.94} +The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results. +[eval step 1] sample: "To solve this problem, first few numbers have been entered into the grid below. Let's start by considering the numbers from $1$ to $49$ arranged in a spiral pattern on a square grid.\n\nThe numbers from" +{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.250888225916028, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 14.3125, "lr": 9e-06, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.6, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.3497639717732866, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 14.5, "lr": 1.2e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 27.5, "frames": {"chat": 207}, "mem_gb": 10.0} +{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.0802153058772286, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 9.3125, "lr": 1.5e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.0, "frames": {"chat": 208}, "mem_gb": 9.96} +{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.9386221677641073, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 6.65625, "lr": 1.8e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8062813847805063, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 4.40625, "lr": 2.1e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 246}, "mem_gb": 9.87} +{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7548754439207415, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 3.484375, "lr": 2.4e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 206}, "mem_gb": 10.02} +{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6572285013991097, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 3.203125, "lr": 2.7000000000000002e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 233}, "mem_gb": 10.01} +{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6164511016746362, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 2.546875, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 229}, "mem_gb": 9.87} +{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5470844178607066, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 1.953125, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 236}, "mem_gb": 9.9} +[eval step 10] sample: "To solve this problem, we need to determine which of the four numbers appear in the shaded squares on the grid.\n\nLet's break down the problem:\n\n1. **Identify the Shaded Squares:**\n The grid is a squ" +{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5313714265652001, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 1.40625, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 239}, "mem_gb": 9.79} +{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4681813031862179, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 1.1953125, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 241}, "mem_gb": 9.91} +{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.49307297485172746, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 1.1171875, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 226}, "mem_gb": 9.87} +{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.44446369409287967, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.97265625, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 234}, "mem_gb": 10.0} +{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4230342352161805, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.90234375, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 257}, "mem_gb": 9.99} +{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5012797900413474, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.99609375, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 208}, "mem_gb": 10.05} +{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4470518503196538, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.90625, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 211}, "mem_gb": 10.02} +{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.41256588681464396, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.83203125, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 227}, "mem_gb": 10.01} +{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.42249245348448555, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.77734375, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 211}, "mem_gb": 9.98} +{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.36867867480044564, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.72265625, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 238}, "mem_gb": 10.0} +[eval step 20] sample: "To solve this problem, we need to analyze the spiral pattern of the numbers from 1 to 49 and determine which four numbers appear in the shaded squares on the same diagonal as the number 7. We'll then " +{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.38083410925380884, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 237}, "mem_gb": 10.04} +{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.39549448682889343, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.765625, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 208}, "mem_gb": 10.04} +{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3769813402572026, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 221}, "mem_gb": 10.12} +{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.35552544313979645, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 3.53125, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 232}, "mem_gb": 9.96} +{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3550635836930325, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.6640625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 208}, "mem_gb": 10.0} +{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.33844906397598484, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.64453125, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 227}, "mem_gb": 9.91} +{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3774184838908414, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.6796875, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 221}, "mem_gb": 9.94} +{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3146386974407981, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 232}, "mem_gb": 10.01} +{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.31184066178612413, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.57421875, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 219}, "mem_gb": 10.01} +{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3350962167671571, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.4, "frames": {"chat": 195}, "mem_gb": 10.1} +[eval step 30] sample: 'To solve this problem, we need to determine the number of prime numbers from 1 to 49 that appear in the shaded squares on a square grid, where the first few numbers are entered into the grid below.\n\n#' +{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.33699155798591673, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.7, "frames": {"chat": 216}, "mem_gb": 10.0} +{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2785733728256077, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.5859375, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.7, "frames": {"chat": 208}, "mem_gb": 9.89} +{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3027569059799115, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.60546875, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 235}, "mem_gb": 9.88} +{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3082514542732388, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.9, "frames": {"chat": 225}, "mem_gb": 9.99} +{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3752551995801429, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.65625, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 213}, "mem_gb": 10.08} +{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2921847006034106, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.85546875, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 257}, "mem_gb": 9.76} +{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3090001767206937, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.546875, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 212}, "mem_gb": 10.03} +{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28328445645694933, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 1.796875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 221}, "mem_gb": 10.0} +{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2893523990919193, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 242}, "mem_gb": 10.0} +{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.255670898507225, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.5546875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 220}, "mem_gb": 9.96} +[eval step 40] sample: 'To solve this problem, we need to analyze the pattern of the numbers from 1 to 49 arranged on a square grid and identify the four numbers that appear in the shaded squares on the same diagonal as the ' +{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2903485574451586, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 240}, "mem_gb": 9.86} +{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28734269070960583, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.6, "frames": {"chat": 206}, "mem_gb": 9.99} +{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27243882406105596, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.52734375, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 226}, "mem_gb": 10.0} +{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3234697731980433, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 244}, "mem_gb": 9.79} +{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3212854475179066, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 224}, "mem_gb": 10.01} +{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28673740869238973, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 271}, "mem_gb": 9.73} +{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2532580826393639, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 1.125, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 236}, "mem_gb": 10.01} +{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28962034405469894, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.55078125, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 232}, "mem_gb": 9.88} +{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23193988344911487, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.4921875, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 210}, "mem_gb": 9.94} +{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22433082110987354, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.90625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.9, "frames": {"chat": 217}, "mem_gb": 9.91} +[eval step 50] sample: 'To solve this problem, we need to analyze the pattern of the numbers from 1 to 49 arranged on a square grid and identify the four numbers that appear in the shaded squares, each appearing on the same ' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep25_s1225/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2995344866629069, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.6171875, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 224}, "mem_gb": 10.03} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32907882408003014, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 203}, "mem_gb": 9.87} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28704522055424747, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 2.4375, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 239}, "mem_gb": 9.97} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23595582261749853, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.57421875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 254}, "mem_gb": 9.88} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21762469755336641, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 241}, "mem_gb": 9.98} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2846538535839568, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 213}, "mem_gb": 10.01} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.245288224790742, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 221}, "mem_gb": 10.05} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2521051613455638, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 1.8046875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.2, "frames": {"chat": 196}, "mem_gb": 10.01} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23354704158020517, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.55078125, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 270}, "mem_gb": 9.82} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19478359519572308, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 216}, "mem_gb": 10.0} +[eval step 60] sample: 'To solve this problem, we need to analyze the pattern of the numbers from 1 to 49 arranged on a square grid and identify the four numbers that appear in the shaded squares, each appearing on the same ' +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.26863892504523196, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.8, "frames": {"chat": 200}, "mem_gb": 9.96} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18590027238409967, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.83984375, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.7, "frames": {"chat": 206}, "mem_gb": 9.91} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2012578780563548, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.474609375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 234}, "mem_gb": 9.95} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.247304932789132, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.3, "frames": {"chat": 215}, "mem_gb": 9.96} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20205045403062055, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.71484375, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 255}, "mem_gb": 9.94} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22750451257377863, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 1.609375, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 250}, "mem_gb": 9.82} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22274810297029715, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 242}, "mem_gb": 10.0} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.26898174686903753, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 3.96875, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2861793573357165, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.6171875, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 208}, "mem_gb": 10.04} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22676605374064918, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 1.1484375, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.4, "frames": {"chat": 208}, "mem_gb": 9.97} +[eval step 70] sample: 'To solve this problem, we need to analyze the spiral pattern of the numbers from 1 to 49 and identify the four numbers that appear in the shaded squares on the same diagonal as the number 7. We will t' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2383458670858294, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 209}, "mem_gb": 10.12} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2040254034437239, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 235}, "mem_gb": 9.96} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18941694195649275, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.490234375, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.5, "frames": {"chat": 204}, "mem_gb": 9.95} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.26322059602600834, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.58984375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20454769928430516, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 1.46875, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 240}, "mem_gb": 10.0} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22102658995253344, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.55078125, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 246}, "mem_gb": 9.99} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22660920705248913, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 243}, "mem_gb": 9.82} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2539451661850015, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.6, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20485700886597236, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.254067884727257, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.6953125, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 211}, "mem_gb": 10.01} +[eval step 80] sample: 'To solve this problem, we need to analyze the spiral pattern of the numbers from 1 to 49 and determine which of the four numbers that appear in the shaded squares on the same diagonal as the number 7 ' +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1860976280965532, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.478515625, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 232}, "mem_gb": 9.98} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21269370155887057, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 214}, "mem_gb": 10.01} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2044230012240509, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 226}, "mem_gb": 9.9} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2102600726918007, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17095083842401704, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.439453125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 218}, "mem_gb": 9.83} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18065360787672302, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23056014474630357, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 215}, "mem_gb": 10.01} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22403676771732667, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1838335989182194, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.0, "frames": {"chat": 209}, "mem_gb": 9.94} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19289504188615827, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 262}, "mem_gb": 9.88} +[eval step 90] sample: "To solve this problem, we need to identify the four numbers that appear in the shaded squares on the same diagonal as the number 7, and then determine how many of these numbers are prime.\n\nHere's the " +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1809839127952854, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 249}, "mem_gb": 9.96} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2376722942231844, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 227}, "mem_gb": 10.0} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.186381211121846, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 221}, "mem_gb": 10.0} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20532299918190886, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 234}, "mem_gb": 10.01} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1591838776408384, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.4, "frames": {"chat": 213}, "mem_gb": 9.96} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16577480874514827, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.6, "frames": {"chat": 213}, "mem_gb": 9.9} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20456980733238161, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 234}, "mem_gb": 9.93} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20402369832278539, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 222}, "mem_gb": 10.0} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21510515817857037, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 227}, "mem_gb": 10.01} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19600455205359807, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 218}, "mem_gb": 10.04} +[eval step 100] sample: 'To solve this problem, we need to:\n\n1. Identify the numbers from 1 to 49 that are arranged in a spiral pattern on a square grid.\n2. Determine which of these numbers are on the same diagonal as the num' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep25_s1225/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2290196806839357, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 223}, "mem_gb": 10.01} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22636315025078754, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.9, "frames": {"chat": 206}, "mem_gb": 10.01} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20191379800935585, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.0, "frames": {"chat": 213}, "mem_gb": 9.92} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2503080016883711, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 223}, "mem_gb": 9.86} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21158150815398744, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 227}, "mem_gb": 9.97} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2013531628333653, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 253}, "mem_gb": 10.0} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21000758766736835, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.9, "frames": {"chat": 216}, "mem_gb": 10.01} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19588753996851543, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.6, "frames": {"chat": 205}, "mem_gb": 9.98} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21358148629758508, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 207}, "mem_gb": 10.06} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19724183553326874, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 215}, "mem_gb": 9.98} +[eval step 110] sample: 'To solve this problem, we need to analyze the spiral pattern of the numbers from 1 to 49 and identify the four numbers that appear on the same diagonal as the number 7. We then determine which of thes' +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14767051264047623, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 228}, "mem_gb": 10.0} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19400057307276874, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.474609375, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 221}, "mem_gb": 10.04} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14793867993044357, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 254}, "mem_gb": 9.84} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12694251759614175, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.9, "frames": {"chat": 210}, "mem_gb": 9.97} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.154613794028759, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 226}, "mem_gb": 9.92} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15858659546474616, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 212}, "mem_gb": 9.99} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17931291315760464, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 211}, "mem_gb": 9.93} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1916892997973909, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.6, "frames": {"chat": 196}, "mem_gb": 9.98} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13504183224166433, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.0, "frames": {"chat": 212}, "mem_gb": 10.0} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14792166811867308, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 244}, "mem_gb": 9.91} +[eval step 120] sample: 'To solve this problem, we need to:\n\n1. Identify the numbers from 1 to 49 arranged in a spiral pattern.\n2. Determine which of these numbers appear on the same diagonal as the number 7.\n3. Count how man' +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14660488209525743, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 222}, "mem_gb": 9.95} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17171216915504386, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 218}, "mem_gb": 10.0} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17044704577376446, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 252}, "mem_gb": 9.88} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18156986663484326, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.7, "frames": {"chat": 202}, "mem_gb": 10.05} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.193661573850736, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 237}, "mem_gb": 10.0} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1703850445672249, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 234}, "mem_gb": 9.99} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12615562903275712, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 215}, "mem_gb": 10.0} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1354192927910636, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 1.0859375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 234}, "mem_gb": 9.93} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12099138240994264, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.7, "frames": {"chat": 216}, "mem_gb": 9.99} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14705536963663374, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.4, "frames": {"chat": 210}, "mem_gb": 9.95} +[eval step 130] sample: 'To solve this problem, we need to analyze the spiral pattern of the numbers from 1 to 49 arranged on a square grid and identify the four numbers that appear on the same diagonal as the number 7. We th' +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1616287318826032, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.4, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14949506662531445, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.0, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15277357547314216, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.4921875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 225}, "mem_gb": 9.95} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15180517331815013, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 253}, "mem_gb": 9.85} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1629465045961241, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 247}, "mem_gb": 9.98} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17035012210526815, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 238}, "mem_gb": 9.98} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19480876584922274, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.7, "frames": {"chat": 235}, "mem_gb": 10.0} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2049359349505355, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.7, "frames": {"chat": 215}, "mem_gb": 9.97} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15441036838572472, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 268}, "mem_gb": 9.97} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17245760981744776, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 228}, "mem_gb": 10.0} +[eval step 140] sample: 'To solve this problem, we need to analyze the spiral pattern of the numbers from 1 to 49 arranged on a square grid and identify the four numbers that appear in the shaded squares, each appearing on th' +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1604549546021968, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 252}, "mem_gb": 9.93} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15443554332548132, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.8, "frames": {"chat": 223}, "mem_gb": 10.01} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19541836336503426, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 226}, "mem_gb": 10.0} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.20092230602238947, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 208}, "mem_gb": 10.05} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1302992971648462, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 240}, "mem_gb": 9.93} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16997946729871133, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 222}, "mem_gb": 9.93} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14522260747464993, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 236}, "mem_gb": 10.0} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13322943811810886, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.0, "frames": {"chat": 217}, "mem_gb": 9.97} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15352233099521448, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.53125, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 252}, "mem_gb": 9.88} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14089195124438653, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 222}, "mem_gb": 9.99} +[eval step 150] sample: 'To solve this problem, we need to:\n\n1. Identify the numbers from 1 to 49 arranged in a spiral pattern.\n2. Determine the positions of the numbers on the same diagonal as the number 7.\n3. Count how many' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep25_s1225/step0150 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len █▃▃▄▄▅▄▅▆▂▃▄▆▄▃▇▄▂▃▆▆▄▅▄▅▅▅▄▆▂▅▄▆▂▅▆▄▂▁▃ +wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▃▃▃▃▄▄▄▄▄▄▄▄▅▅▅▅▅▅▅▅▆▆▆▆▇▇▇▇▇███ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅███████████ +wandb: finish_rate ▂▅▃▁▆▅▃▅▅▃▇▂█▆▄█▅▂▃▅▄▅▄▅▄▂▄▇▅▄▅▃▆▄▄▅▆▅▄▆ +wandb: forward_topk_kl █▆▅▄▄▃▃▃▂▃▂▂▂▂▂▂▂▂▂▂▂▂▁▂▂▂▂▂▂▁▁▁▁▁▁▁▂▁▁▁ +wandb: grad_norm █▅▄▂▂▁▁▁▃▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▇█████████████████████████████████████ +wandb: mem_gb ▅▇▇▆▇▆▆▅▇▆█▂▇▄▁▅▅▆▆▅▃▇▆▇▆▆▇▄▆▆▅▆▆▅▆▅▃▆▆▆ +wandb: step ▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▆▇▇▇▇▇▇█ +wandb: t_data_s █▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 540.5 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.847 +wandb: forward_topk_kl 0.14089 +wandb: grad_norm 0.38281 +wandb: lr 3e-05 +wandb: mem_gb 9.99 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run uniform-math-keep25-s1225 at: https://wandb.ai/hbfreed/glean-grid/runs/a8l8mpfr +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/uniform_keep25_s1225/wandb/run-20260716_051726-a8l8mpfr/logs +{ + "correct": 267, + "accuracy": 0.20242608036391205, + "finished": 1270, + "finish_rate": 0.9628506444275967, + "mean_completion_tokens": 150.4215314632297 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep25_s1225_step100_chat.json +{ + "correct": 291, + "accuracy": 0.22062168309325247, + "finished": 1272, + "finish_rate": 0.9643669446550417, + "mean_completion_tokens": 148.76952236542834 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep25_s1225_step150_chat.json diff --git a/healed/grid_math/uniform_keep25_s1226.console.log b/healed/grid_math/uniform_keep25_s1226.console.log new file mode 100644 index 0000000000000000000000000000000000000000..8ac8742563d90c65eaca19d2549505aae939c77d --- /dev/null +++ b/healed/grid_math/uniform_keep25_s1226.console.log @@ -0,0 +1,232 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run nzko6jqq +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/uniform_keep25_s1226/wandb/run-20260716_051619-nzko6jqq +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run uniform-math-keep25-s1226 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/nzko6jqq +12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 2.09B | teacher overlap=False +{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2495578979462385, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 14.5625, "lr": 6e-06, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.2, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 254}, "mem_gb": 9.82} +The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results. +[eval step 1] sample: "The perimeter of a triangle is given as 28, and the midpoints of its sides are connected by segments. Let's denote the sides of the triangle as follows: a, b, and c. The midpoints of the sides are con" +{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2842795796026787, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 14.25, "lr": 9e-06, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 241}, "mem_gb": 9.98} +{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2860378812308113, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 13.3125, "lr": 1.2e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.7, "frames": {"chat": 213}, "mem_gb": 10.01} +{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.0296244965081414, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 8.625, "lr": 1.5e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 221}, "mem_gb": 10.05} +{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.0173526370272041, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 6.9375, "lr": 1.8e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.0, "frames": {"chat": 196}, "mem_gb": 10.01} +{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7652708411070208, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 4.1875, "lr": 2.1e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 270}, "mem_gb": 9.82} +{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7675839649726948, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 3.734375, "lr": 2.4e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.0, "frames": {"chat": 216}, "mem_gb": 10.0} +{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7070816349441806, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 3.28125, "lr": 2.7000000000000002e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.7, "frames": {"chat": 200}, "mem_gb": 9.96} +{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6169270259405176, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 2.75, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.5, "frames": {"chat": 206}, "mem_gb": 9.91} +{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5323196953435739, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 1.9609375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 234}, "mem_gb": 9.95} +[eval step 10] sample: "To find the perimeter of a triangle formed by connecting the midpoints of its sides, we need to determine the length of each of the three sides of the triangle.\n\nLet's break down the problem step-by-s" +{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5534217075144251, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 1.640625, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.2, "frames": {"chat": 215}, "mem_gb": 9.96} +{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.470129220243295, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 1.1953125, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 255}, "mem_gb": 9.94} +{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4660646371759474, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 1.09375, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 250}, "mem_gb": 9.82} +{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4519369126672546, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 1.0234375, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 242}, "mem_gb": 10.0} +{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.48208508322536947, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.97265625, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.1, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.505666505916665, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.98046875, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 208}, "mem_gb": 10.04} +{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.42821578803832333, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 1.0078125, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.3, "frames": {"chat": 208}, "mem_gb": 9.97} +{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4488865850756566, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.8359375, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 209}, "mem_gb": 10.12} +{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3713087936893105, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 12.4375, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 235}, "mem_gb": 9.96} +{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3715890112740298, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.80078125, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.4, "frames": {"chat": 204}, "mem_gb": 9.95} +[eval step 20] sample: 'To solve this problem, we need to understand the geometry of the triangle and the properties of its midpoints. The perimeter of a triangle is given by the sum of its sides, and the midpoints of its si' +{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.44148291802903017, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.9375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.36412479583832125, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.80859375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 240}, "mem_gb": 10.0} +{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.35719255141379935, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.69921875, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 246}, "mem_gb": 9.99} +{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3621041040317466, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 243}, "mem_gb": 9.82} +{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.38283104023163517, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.68359375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.5, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3491992857553065, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.1, "frames": {"chat": 219}, "mem_gb": 10.0} +{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3949156841669232, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.80859375, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 211}, "mem_gb": 10.01} +{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3246795868317286, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.69140625, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 232}, "mem_gb": 9.98} +{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3552298259820789, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.73046875, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.1, "frames": {"chat": 214}, "mem_gb": 10.01} +{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3321069305093338, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.65625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.7, "frames": {"chat": 226}, "mem_gb": 9.9} +[eval step 30] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the connections between its midpoints.\n\n1. **Understand the Geometry:**\n - The perimeter of a triangle is gi' +{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32228971752213936, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.64453125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2798056041251868, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 218}, "mem_gb": 9.83} +{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.29189739949926735, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.35060337477376063, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.64453125, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 215}, "mem_gb": 10.01} +{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.33954473176573713, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 233}, "mem_gb": 9.99} +{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2843651156159739, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 1.828125, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.9, "frames": {"chat": 209}, "mem_gb": 9.94} +{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2947497258403649, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.72265625, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 262}, "mem_gb": 9.88} +{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28674538816077016, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 249}, "mem_gb": 9.96} +{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.33914187191314993, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 1.4921875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 227}, "mem_gb": 10.0} +{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.26380810491734497, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 221}, "mem_gb": 10.0} +[eval step 40] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the connections between its midpoints.\n\n1. **Understand the Geometry:**\n - The perimeter \\( P \\) of a triang' +{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.302417300863564, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.5546875, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 234}, "mem_gb": 10.01} +{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24419383710200587, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.52734375, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.3, "frames": {"chat": 213}, "mem_gb": 9.96} +{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24537141441050916, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.5, "frames": {"chat": 213}, "mem_gb": 9.9} +{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27504100236129014, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.53125, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 234}, "mem_gb": 9.93} +{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2705650013284758, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 222}, "mem_gb": 10.0} +{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.31124431811695297, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.59375, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 227}, "mem_gb": 10.01} +{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2672547916886707, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 218}, "mem_gb": 10.04} +{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3164061588189254, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.57421875, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 223}, "mem_gb": 10.01} +{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.31090367152442533, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.8, "frames": {"chat": 206}, "mem_gb": 10.01} +{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2863034582992395, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.5625, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.9, "frames": {"chat": 213}, "mem_gb": 9.92} +[eval step 50] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the connections between its midpoints.\n\n1. **Understand the Perimeter:**\n The perimeter \\( P \\) of a triangl' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep25_s1226/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.33238482810370623, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.5859375, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 223}, "mem_gb": 9.86} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.30147026825944584, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 227}, "mem_gb": 9.97} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2694238721400499, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 1.2109375, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 253}, "mem_gb": 10.0} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2599986071868489, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.7, "frames": {"chat": 216}, "mem_gb": 10.01} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.26703055294280253, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.53125, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.4, "frames": {"chat": 205}, "mem_gb": 9.98} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2845016695648432, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.6328125, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 207}, "mem_gb": 10.06} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2710578287235151, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 215}, "mem_gb": 9.98} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2025449087051054, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 228}, "mem_gb": 10.0} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2637373869329691, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.5625, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 221}, "mem_gb": 10.04} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20766283342484385, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 254}, "mem_gb": 9.84} +[eval step 60] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the connections between its midpoints.\n\n1. **Understand the Geometry:**\n - Let the sides of the triangle be ' +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17684483035219212, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.7, "frames": {"chat": 210}, "mem_gb": 9.97} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21151079111825674, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 226}, "mem_gb": 9.92} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22429327983123562, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.1, "frames": {"chat": 212}, "mem_gb": 9.99} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23901226744391024, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.4921875, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 211}, "mem_gb": 9.93} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2521503586698324, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.53125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.5, "frames": {"chat": 196}, "mem_gb": 9.98} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19568856958678613, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.8, "frames": {"chat": 212}, "mem_gb": 10.0} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2056147424393023, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 244}, "mem_gb": 9.91} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19970010116541137, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 222}, "mem_gb": 9.95} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2406689625217269, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.1, "frames": {"chat": 218}, "mem_gb": 10.0} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2351806751595189, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 252}, "mem_gb": 9.88} +[eval step 70] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the connections between its midpoints.\n\n1. **Understand the Geometry:**\n - Let the sides of the triangle be ' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23980661761003236, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.5, "frames": {"chat": 202}, "mem_gb": 10.05} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2756894440931578, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.59765625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 237}, "mem_gb": 10.0} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22606769043393432, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 234}, "mem_gb": 9.99} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17098218618429575, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 215}, "mem_gb": 10.0} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18496631520005563, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.7, "frames": {"chat": 234}, "mem_gb": 9.93} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16087721765795723, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.6, "frames": {"chat": 216}, "mem_gb": 9.99} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1994732022792101, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.5, "frames": {"chat": 210}, "mem_gb": 9.95} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20232079331736702, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.2, "frames": {"chat": 199}, "mem_gb": 10.0} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18569014142416418, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.9, "frames": {"chat": 210}, "mem_gb": 10.01} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20507156935961296, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.1, "frames": {"chat": 225}, "mem_gb": 9.95} +[eval step 80] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the connections between its midpoints.\n\n1. **Understand the Problem:**\n - We are given the perimeter of the ' +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20882732380144298, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 253}, "mem_gb": 9.85} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22202617851061127, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 247}, "mem_gb": 9.98} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2211942642432948, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 238}, "mem_gb": 9.98} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23084418179870894, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 235}, "mem_gb": 10.0} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2283516202347974, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 215}, "mem_gb": 9.97} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22033791828608762, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 268}, "mem_gb": 9.97} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22125197298427424, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 228}, "mem_gb": 10.0} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21613482079487295, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 252}, "mem_gb": 9.93} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1895898001347358, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.6, "frames": {"chat": 223}, "mem_gb": 10.01} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.26797097403767206, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.55078125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 226}, "mem_gb": 10.0} +[eval step 90] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Problem:**\n - We are given the perimeter of a triangle, which is 28.\n - The midpoints of the sides of the triangle are co' +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24367389039595921, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.55078125, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 208}, "mem_gb": 10.05} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17548983993784836, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 240}, "mem_gb": 9.93} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21608840946884206, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 222}, "mem_gb": 9.93} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1903983515452904, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 236}, "mem_gb": 10.0} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17124010507706552, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.8, "frames": {"chat": 217}, "mem_gb": 9.97} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21378996573444456, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.59375, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 252}, "mem_gb": 9.88} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17583670812180888, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.1, "frames": {"chat": 222}, "mem_gb": 9.99} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2008310172341764, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.901, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 242}, "mem_gb": 9.87} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2504596530464788, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 219}, "mem_gb": 9.93} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20776792142900327, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.8, "frames": {"chat": 223}, "mem_gb": 9.94} +[eval step 100] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the connections between its midpoints.\n\n1. **Understand the Problem:**\n - We are given the perimeter of the ' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep25_s1226/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21407153125032782, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 232}, "mem_gb": 9.95} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22760161069128662, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.53125, "lr": 3e-05, "finish_rate": 0.832, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 220}, "mem_gb": 10.0} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.23536558933630586, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 210}, "mem_gb": 10.04} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20387262995596975, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 226}, "mem_gb": 9.97} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1928112811392794, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.741, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.8, "frames": {"chat": 212}, "mem_gb": 10.0} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18794621144427606, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.48828125, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 236}, "mem_gb": 10.01} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16627150794851284, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.928, "comp_len": 454.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 264}, "mem_gb": 9.88} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21027346796846638, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 229}, "mem_gb": 9.98} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16645300974740337, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 465.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 258}, "mem_gb": 9.85} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21918750831211606, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 208}, "mem_gb": 10.01} +[eval step 110] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the connections between its midpoints.\n\n1. **Understand the Geometry:**\n - Let the sides of the triangle be ' +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18296561656640842, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.88, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 249}, "mem_gb": 9.93} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14481682858659575, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.5, "frames": {"chat": 220}, "mem_gb": 9.99} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1859534092683966, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.8, "frames": {"chat": 223}, "mem_gb": 9.99} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13867307858555578, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.1, "frames": {"chat": 221}, "mem_gb": 10.0} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15031419390533118, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 230}, "mem_gb": 9.9} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17688426749395827, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 214}, "mem_gb": 9.97} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2198834235218664, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 214}, "mem_gb": 9.99} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1655972873053203, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.478515625, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 210}, "mem_gb": 10.04} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18060686992689345, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 214}, "mem_gb": 10.0} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.18515299578898897, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.791, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 215}, "mem_gb": 9.96} +[eval step 120] sample: "To solve this problem, we need to understand the geometric properties involved. Here's a step-by-step breakdown:\n\n1. **Understand the Problem:**\n - We have a triangle with sides \\(a\\), \\(b\\), and \\(" +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19961161344274878, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.4921875, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 208}, "mem_gb": 9.99} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.150687341630583, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.789, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 218}, "mem_gb": 9.88} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1382546880559375, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.7, "frames": {"chat": 233}, "mem_gb": 9.9} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13145577659346164, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 231}, "mem_gb": 9.94} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16622866188141827, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 235}, "mem_gb": 10.13} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17218660681688538, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.1, "frames": {"chat": 216}, "mem_gb": 9.99} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16202461753959457, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.831, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 237}, "mem_gb": 10.01} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1946208799743404, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 203}, "mem_gb": 10.01} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16476607479564845, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.873, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 236}, "mem_gb": 10.04} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16959726118591303, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.0, "frames": {"chat": 214}, "mem_gb": 10.01} +[eval step 130] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the connections between its midpoints.\n\n1. **Understanding the Problem:**\n - We are given the perimeter of a' +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15791527282614262, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.7, "frames": {"chat": 209}, "mem_gb": 10.0} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16019434139728547, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 256}, "mem_gb": 10.0} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1639191758962348, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 230}, "mem_gb": 9.87} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16772623434948425, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.439453125, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 230}, "mem_gb": 10.06} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1819166350507488, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 227}, "mem_gb": 9.95} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1837459068759655, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.7, "frames": {"chat": 208}, "mem_gb": 10.01} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.19119155149323244, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.474609375, "lr": 3e-05, "finish_rate": 0.699, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 206}, "mem_gb": 10.03} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17735388877652586, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.82, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 228}, "mem_gb": 9.9} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17459255363605916, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 224}, "mem_gb": 10.0} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15427039775537948, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.66, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.1, "frames": {"chat": 200}, "mem_gb": 10.03} +[eval step 140] sample: "To solve this problem, we need to understand the geometric properties involved. Let's break down the problem step-by-step:\n\n1. **Understand the Geometry:**\n - A triangle has three sides, say \\(a\\), " +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17032653220028926, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.714, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.8, "frames": {"chat": 196}, "mem_gb": 10.01} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14531434423498188, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 223}, "mem_gb": 10.0} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13336851223480578, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.869, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.4, "frames": {"chat": 213}, "mem_gb": 9.89} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14154418901971852, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 232}, "mem_gb": 9.93} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15019642852085333, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 223}, "mem_gb": 9.93} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.15932164447531105, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.85, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.7, "frames": {"chat": 233}, "mem_gb": 10.02} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16258757215073952, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.816, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.9, "frames": {"chat": 217}, "mem_gb": 10.01} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.21068223652498175, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.752, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 202}, "mem_gb": 10.08} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14553914751367023, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 253}, "mem_gb": 9.93} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13818583363940318, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 231}, "mem_gb": 9.94} +[eval step 150] sample: "To solve this problem, we need to understand the geometric properties involved. Let's break down the problem into manageable steps:\n\n1. **Understand the Geometry:**\n - A triangle has a given perimet" +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep25_s1226/step0150 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: uploading history steps 149-149, summary, console lines 168-170 +wandb: +wandb: Run history: +wandb: comp_len ▅▂▆▇▃▃▆▄▄▁▄▅▄▆▅▇▆▄█▃▅▂▂▅▅▆▄▃▁▁▄▆▆▆▆▆▆▅▇▅ +wandb: cumulative_loss_tokens ▁▁▁▂▂▂▂▂▂▃▃▄▄▄▄▄▄▅▅▅▅▅▅▅▅▆▆▆▆▆▆▆▇▇▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅██████████ +wandb: finish_rate █▇▃█▃▅▅▆▄▆▅▅▆█▇▄▅▇█▅▇▆▇▆▇▇▆▃▆▄█▄▅▄▅▇▆▆▁▆ +wandb: forward_topk_kl █▅▄▄▃▃▂▂▂▂▂▂▂▂▂▂▂▂▂▂▂▂▂▂▂▂▂▂▂▁▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm █▃▂▂▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂██████████████████████████████████████ +wandb: mem_gb ▄▄▄▇▆▆▆▄▄▃▃▆▆▆▅▄▇▆▄▆▅▅▆▆▆▃▅▆▂▁▃▅▅▆▄▄▆▆▂█ +wandb: step ▁▁▁▁▂▂▂▂▂▃▃▃▃▄▄▄▄▄▄▄▅▅▅▅▅▆▆▆▇▇▇▇▇▇▇▇████ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 519.5 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.879 +wandb: forward_topk_kl 0.13819 +wandb: grad_norm 0.39648 +wandb: lr 3e-05 +wandb: mem_gb 9.94 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run uniform-math-keep25-s1226 at: https://wandb.ai/hbfreed/glean-grid/runs/nzko6jqq +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/uniform_keep25_s1226/wandb/run-20260716_051619-nzko6jqq/logs +{ + "correct": 264, + "accuracy": 0.2001516300227445, + "finished": 1257, + "finish_rate": 0.9529946929492039, + "mean_completion_tokens": 155.3972706595906 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep25_s1226_step100_chat.json +{ + "correct": 300, + "accuracy": 0.22744503411675512, + "finished": 1273, + "finish_rate": 0.9651250947687642, + "mean_completion_tokens": 150.46095526914328 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep25_s1226_step150_chat.json diff --git a/healed/grid_math/uniform_keep50_s1224.console.log b/healed/grid_math/uniform_keep50_s1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..680f13ddf6ec035bc75c4823d4e6b98fded59709 --- /dev/null +++ b/healed/grid_math/uniform_keep50_s1224.console.log @@ -0,0 +1,232 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run r9qva0v2 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/uniform_keep50_s1224/wandb/run-20260716_000817-r9qva0v2 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run uniform-math-keep50-s1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/r9qva0v2 + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/grid_math/uniform_keep50_s1224/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13515835460849726, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 222}, "mem_gb": 16.0} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.11656722308822597, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.9, "frames": {"chat": 235}, "mem_gb": 16.05} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13514603171708683, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.7, "frames": {"chat": 208}, "mem_gb": 16.01} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12384081650370111, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.3, "frames": {"chat": 191}, "mem_gb": 16.05} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09931092557078228, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 219}, "mem_gb": 16.05} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09109834918997561, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.4, "frames": {"chat": 207}, "mem_gb": 16.05} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11065167818926275, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 208}, "mem_gb": 16.01} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08927264785487204, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.8, "frames": {"chat": 219}, "mem_gb": 16.04} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08789662679632505, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 246}, "mem_gb": 15.92} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11347239079214633, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 206}, "mem_gb": 16.07} +[eval step 60] sample: "To solve the given system of equations, we will use Python and SymPy to handle the algebraic manipulations. Let's break down the problem step-by-step and write the Python code to find the value of \\( " +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0887840253782769, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 233}, "mem_gb": 16.05} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09239683715384453, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.5, "frames": {"chat": 229}, "mem_gb": 15.92} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08232894674045965, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 236}, "mem_gb": 15.95} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09751218746158605, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 239}, "mem_gb": 15.84} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08627100243655343, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 241}, "mem_gb": 15.96} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09800913436881577, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.7, "frames": {"chat": 226}, "mem_gb": 15.92} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08852198531199247, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 234}, "mem_gb": 16.05} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08382236090765025, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 257}, "mem_gb": 16.04} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12105931493146345, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 208}, "mem_gb": 16.1} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1038044841277413, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.3, "frames": {"chat": 211}, "mem_gb": 16.07} +[eval step 70] sample: 'To solve the given system of equations, we will use Python and SymPy to handle the algebraic manipulations. The equations are:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb + m + r &=' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1100768730642274, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.8, "frames": {"chat": 227}, "mem_gb": 16.05} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10338081272356212, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 211}, "mem_gb": 16.03} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08453753360584378, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 238}, "mem_gb": 16.04} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09178873166749253, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 237}, "mem_gb": 16.08} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10419285486250494, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.1, "frames": {"chat": 208}, "mem_gb": 16.08} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09346181132777905, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 221}, "mem_gb": 16.17} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08947489535867546, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 232}, "mem_gb": 16.01} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09227189582098896, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.6, "frames": {"chat": 208}, "mem_gb": 16.04} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08777518702357387, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 227}, "mem_gb": 15.96} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10104774889331311, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.0, "frames": {"chat": 221}, "mem_gb": 15.99} +[eval step 80] sample: 'To solve the given system of equations, we need to express \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) in terms of each other and solve for their values.\n\nThe system of equations is:\n\\[\n\\begin{align*' +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08170304526534553, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.8, "frames": {"chat": 232}, "mem_gb": 16.05} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08983821267243475, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.6, "frames": {"chat": 219}, "mem_gb": 16.05} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09457728577253098, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.6, "frames": {"chat": 195}, "mem_gb": 16.14} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09356701532015577, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 216}, "mem_gb": 16.05} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09165108890738338, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.8, "frames": {"chat": 208}, "mem_gb": 15.94} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08849761248029148, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 235}, "mem_gb": 15.93} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08910521061414232, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.3, "frames": {"chat": 225}, "mem_gb": 16.04} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11201870662448928, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.0, "frames": {"chat": 213}, "mem_gb": 16.13} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07798361969428758, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "t_data_s": 0.2, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 257}, "mem_gb": 15.8} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09493755235892411, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.1, "frames": {"chat": 212}, "mem_gb": 16.08} +[eval step 90] sample: "To solve the given system of equations, we need to express each variable \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) in terms of each other using the given equations. Let's break down the problem ste" +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08449767493788773, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.6, "frames": {"chat": 221}, "mem_gb": 16.05} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08178153519337066, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 242}, "mem_gb": 16.04} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07604952207386183, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.7, "frames": {"chat": 220}, "mem_gb": 16.01} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0801545599025519, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.2, "frames": {"chat": 240}, "mem_gb": 15.9} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08843556518893068, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.8, "frames": {"chat": 206}, "mem_gb": 16.04} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08580168459989751, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.9, "frames": {"chat": 226}, "mem_gb": 16.05} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09918145173918456, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 244}, "mem_gb": 15.84} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10544447463347266, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 224}, "mem_gb": 16.05} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08442806187899163, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 271}, "mem_gb": 15.78} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0845842156385382, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 236}, "mem_gb": 16.06} +[eval step 100] sample: "To solve the given system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) such that each letter represents a non-zero digit. Let's break down the problem " +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep50_s1224/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08446701570067865, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.9, "frames": {"chat": 232}, "mem_gb": 15.93} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0794957819807964, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.6, "frames": {"chat": 210}, "mem_gb": 15.99} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07818544425293804, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.7, "frames": {"chat": 217}, "mem_gb": 15.95} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09452820008428146, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 224}, "mem_gb": 16.07} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11034645940602447, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 203}, "mem_gb": 15.92} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08894736041029294, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.0, "frames": {"chat": 239}, "mem_gb": 16.02} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06482621920686216, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 254}, "mem_gb": 15.93} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0596997441777649, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 241}, "mem_gb": 16.03} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08630636698001375, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 213}, "mem_gb": 16.05} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07119898068271577, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 221}, "mem_gb": 16.1} +[eval step 110] sample: "To solve the given system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) such that each letter represents a non-zero digit. Let's break down the problem " +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07626933450518797, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.2, "frames": {"chat": 196}, "mem_gb": 16.06} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.062216147240251304, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 270}, "mem_gb": 15.86} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06098471744398897, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 216}, "mem_gb": 16.04} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07360812601515403, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.6, "frames": {"chat": 200}, "mem_gb": 16.01} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06202772317926089, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.5, "frames": {"chat": 206}, "mem_gb": 15.96} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05959480526245509, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 234}, "mem_gb": 15.99} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07201705848282824, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.3, "frames": {"chat": 215}, "mem_gb": 16.01} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0571913227643352, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 255}, "mem_gb": 15.99} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05931704173743104, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 250}, "mem_gb": 15.87} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06392476364959342, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.6, "frames": {"chat": 242}, "mem_gb": 16.04} +[eval step 120] sample: "To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) such that the given equations hold true. Let's break down the problem step-by-step:\n\n1. **Define V" +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07929505948486427, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.3, "frames": {"chat": 199}, "mem_gb": 16.05} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08340449355278785, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 208}, "mem_gb": 16.08} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06475446787502151, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 208}, "mem_gb": 16.02} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0749057637304999, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.3, "frames": {"chat": 209}, "mem_gb": 16.17} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05738816611807172, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 235}, "mem_gb": 16.01} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0597335497791258, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.2, "frames": {"chat": 204}, "mem_gb": 16.0} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08213897060084467, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.0, "frames": {"chat": 208}, "mem_gb": 16.05} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05984125637561083, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 240}, "mem_gb": 16.05} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0629925194332997, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 246}, "mem_gb": 16.04} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06474258221313357, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 243}, "mem_gb": 15.86} +[eval step 130] sample: "To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) such that the given equations hold true. Let's break down the problem step-by-step and use Python " +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07210097291904191, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.2, "frames": {"chat": 208}, "mem_gb": 16.06} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06917726617272323, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.7, "frames": {"chat": 219}, "mem_gb": 16.05} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08181537830297214, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.0, "frames": {"chat": 211}, "mem_gb": 16.06} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06436642667967826, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.8, "frames": {"chat": 232}, "mem_gb": 16.02} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06862061261056612, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.8, "frames": {"chat": 214}, "mem_gb": 16.05} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06350242085993911, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.8, "frames": {"chat": 226}, "mem_gb": 15.95} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06530722830637048, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 210}, "mem_gb": 16.06} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05820518815023824, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.9, "frames": {"chat": 218}, "mem_gb": 15.88} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05650871475211655, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.2431640625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 233}, "mem_gb": 16.04} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07226877674786374, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.1, "frames": {"chat": 215}, "mem_gb": 16.05} +[eval step 140] sample: "To solve the given system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) such that each letter represents a non-zero digit. Let's break down the problem " +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06750662778724606, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.2, "frames": {"chat": 233}, "mem_gb": 16.04} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.061347212908416986, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.9, "frames": {"chat": 209}, "mem_gb": 15.99} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.055460787860878435, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.0, "frames": {"chat": 262}, "mem_gb": 15.92} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.058363437791649875, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.6, "frames": {"chat": 249}, "mem_gb": 16.01} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07328259567250497, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.2, "frames": {"chat": 227}, "mem_gb": 16.05} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.058721957165344306, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.0, "frames": {"chat": 221}, "mem_gb": 16.04} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.061519284149631856, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.2, "frames": {"chat": 234}, "mem_gb": 16.06} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.056329902307611576, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.0, "frames": {"chat": 213}, "mem_gb": 16.0} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05893662882659895, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.5, "frames": {"chat": 213}, "mem_gb": 15.94} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05945110498759896, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.1, "frames": {"chat": 234}, "mem_gb": 15.97} +[eval step 150] sample: "To solve the given system of equations, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) such that each letter represents a non-zero digit. Let's break down the problem step-" +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep50_s1224/step0150 +wandb: updating run metadata +wandb: uploading data +wandb: +wandb: Run history: +wandb: comp_len ▆▆▄▄▄▃▅▃▃▄▆▅▄█▆▆▄▄▄▆▄▅█▅▄▃▃▆▅▁▂▁▆▂▆▆▆▅▅▂ +wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▂▂▃▃▃▃▄▄▄▅▅▅▅▅▅▅▆▆▆▆▆▆▆▇▇▇▇████ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅████████████ +wandb: finish_rate ▅▅▄▆▆▆▅▅▆▃▁▆▆▆▆▄▃▅▁▄▆█▅▆▇█▅▄▅▆▅▇▇▂▃▂▂█▅▅ +wandb: forward_topk_kl █▆▄▄▄▄▄▄▃▃▃▃▂▃▂▂▂▃▂▂▂▃▂▂▂▂▂▂▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm █▇▄▃▂▂▂▂▂▂▂▂▁▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▆█████████████████████████████████████ +wandb: mem_gb ▅▆▆▆▆▄▅▆▆▆▇▆▆▃▃▂▃▅▆▆█▅▆▆▁▃▆▆▃▄▅▅▂▆▅▆▆▆▅▄ +wandb: step ▁▁▁▁▁▂▂▂▂▃▃▃▃▃▃▃▃▄▄▄▄▄▄▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇▇█ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 512.8 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.906 +wandb: forward_topk_kl 0.05945 +wandb: grad_norm 0.35156 +wandb: lr 3e-05 +wandb: mem_gb 15.97 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run uniform-math-keep50-s1224 at: https://wandb.ai/hbfreed/glean-grid/runs/r9qva0v2 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/uniform_keep50_s1224/wandb/run-20260716_000817-r9qva0v2/logs +{ + "correct": 668, + "accuracy": 0.5064442759666414, + "finished": 1306, + "finish_rate": 0.9901440485216073, + "mean_completion_tokens": 111.20773313115997 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep50_s1224_step100_chat.json +{ + "correct": 683, + "accuracy": 0.5178165276724791, + "finished": 1312, + "finish_rate": 0.9946929492039424, + "mean_completion_tokens": 109.86429112964368 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep50_s1224_step150_chat.json diff --git a/healed/grid_math/uniform_keep50_s1225.console.log b/healed/grid_math/uniform_keep50_s1225.console.log new file mode 100644 index 0000000000000000000000000000000000000000..5f31ca822d9fbdc4879ec5a15df8ae5867c09266 --- /dev/null +++ b/healed/grid_math/uniform_keep50_s1225.console.log @@ -0,0 +1,231 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/uniform_keep50_s1225/wandb/run-20260716_000817-tj9y7tlh +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run uniform-math-keep50-s1225 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/tj9y7tlh + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/grid_math/uniform_keep50_s1225/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.14507852032749602, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 224}, "mem_gb": 16.07} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16389974264514942, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 203}, "mem_gb": 15.92} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1358009697439149, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.2, "frames": {"chat": 239}, "mem_gb": 16.02} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0977568897123759, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.9, "frames": {"chat": 254}, "mem_gb": 15.93} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0903720636972847, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.1, "frames": {"chat": 241}, "mem_gb": 16.03} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1253400321662426, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.5, "frames": {"chat": 213}, "mem_gb": 16.05} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1029980695111677, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.7, "frames": {"chat": 221}, "mem_gb": 16.1} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10780552417521054, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.0, "frames": {"chat": 196}, "mem_gb": 16.06} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09320413307485481, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.1, "frames": {"chat": 270}, "mem_gb": 15.86} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08679414614799122, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 216}, "mem_gb": 16.04} +[eval step 60] sample: 'To solve this problem, we need to identify the four numbers from the sequence \\(1, 2, 3, 4, 5, 6, 7, \\ldots, 49\\) that lie on the same diagonal as the number \\(7\\) and are prime.\n\n### Steps to Solve t' +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12127362799135347, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.6, "frames": {"chat": 200}, "mem_gb": 16.01} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08365503143308063, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 206}, "mem_gb": 15.96} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08216556253097951, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.7, "frames": {"chat": 234}, "mem_gb": 15.99} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10262446160347512, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.3, "frames": {"chat": 215}, "mem_gb": 16.01} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08079534837898487, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.8, "frames": {"chat": 255}, "mem_gb": 15.99} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08919197535691782, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.8, "frames": {"chat": 250}, "mem_gb": 15.87} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09303663084845369, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.3, "frames": {"chat": 242}, "mem_gb": 16.04} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12093032057676464, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 199}, "mem_gb": 16.05} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1223956153758491, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.9, "frames": {"chat": 208}, "mem_gb": 16.08} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09326641459406043, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.2, "frames": {"chat": 208}, "mem_gb": 16.02} +[eval step 70] sample: 'To solve this problem, we need to understand the structure of the spiral pattern on the grid and identify the numbers that lie on the same diagonal as the number \\(7\\).\n\n### Steps to Solve the Problem' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1034547137827302, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.9, "frames": {"chat": 209}, "mem_gb": 16.17} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08132403246794517, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 235}, "mem_gb": 16.01} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08316841166916614, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.4, "frames": {"chat": 204}, "mem_gb": 16.0} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11559949020318067, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.9, "frames": {"chat": 208}, "mem_gb": 16.05} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08412509058462456, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.9, "frames": {"chat": 240}, "mem_gb": 16.05} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09016936350489656, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.0, "frames": {"chat": 246}, "mem_gb": 16.04} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0944399022025677, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.9, "frames": {"chat": 243}, "mem_gb": 15.86} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10787633103836948, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.5, "frames": {"chat": 208}, "mem_gb": 16.06} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09065167806303749, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 219}, "mem_gb": 16.05} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1123543589844058, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.4, "frames": {"chat": 211}, "mem_gb": 16.06} +[eval step 80] sample: "To solve this problem, we need to identify the four numbers on the same diagonal as the number \\(7\\) in a spiral pattern of numbers from 1 to 49 arranged on a square grid. Let's break down the steps:\n" +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08160814283859606, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.4, "frames": {"chat": 232}, "mem_gb": 16.02} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0911076406781872, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 214}, "mem_gb": 16.05} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09021442107763142, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 226}, "mem_gb": 15.95} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09332444871294622, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 210}, "mem_gb": 16.06} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07827468976057135, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.3046875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 218}, "mem_gb": 15.88} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07789079143583465, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.1, "frames": {"chat": 233}, "mem_gb": 16.04} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1009345404283454, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.4, "frames": {"chat": 215}, "mem_gb": 16.05} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0944027965891175, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 233}, "mem_gb": 16.04} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08136886405814439, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.9, "frames": {"chat": 209}, "mem_gb": 15.99} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07810565605591982, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.5, "frames": {"chat": 262}, "mem_gb": 15.92} +[eval step 90] sample: 'To solve this problem, we need to identify the four numbers on the diagonal that correspond to the number \\(7\\) in the spiral pattern of the grid. The spiral pattern starts at the center and moves clo' +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07821270908623312, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.5, "frames": {"chat": 249}, "mem_gb": 16.01} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10085701854179303, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.7, "frames": {"chat": 227}, "mem_gb": 16.05} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0847742678733853, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 221}, "mem_gb": 16.04} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08609593580262735, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 234}, "mem_gb": 16.06} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07459482868239284, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.2, "frames": {"chat": 213}, "mem_gb": 16.0} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07537244438646983, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.4, "frames": {"chat": 213}, "mem_gb": 15.94} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08689684214126318, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.1, "frames": {"chat": 234}, "mem_gb": 15.97} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09049659343931514, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 222}, "mem_gb": 16.04} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09133215679507703, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.3, "frames": {"chat": 227}, "mem_gb": 16.06} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08776675813384355, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 218}, "mem_gb": 16.09} +[eval step 100] sample: 'To solve this problem, we need to understand the structure of the spiral pattern on the grid and how the numbers are arranged. The numbers from 1 to 49 are arranged in a spiral pattern, starting from ' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep50_s1225/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0949454252282468, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 223}, "mem_gb": 16.06} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09777595625078926, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.9, "frames": {"chat": 206}, "mem_gb": 16.05} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08421979916729033, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 213}, "mem_gb": 15.97} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11003809169378753, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.7, "frames": {"chat": 223}, "mem_gb": 15.91} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08955485637433205, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 227}, "mem_gb": 16.02} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07951079289832463, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.0, "frames": {"chat": 253}, "mem_gb": 16.05} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08083365453135533, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.8, "frames": {"chat": 216}, "mem_gb": 16.06} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07531270161882664, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.5, "frames": {"chat": 205}, "mem_gb": 16.02} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08617055136881148, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.7, "frames": {"chat": 207}, "mem_gb": 16.11} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07958229148842705, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.1, "frames": {"chat": 215}, "mem_gb": 16.03} +[eval step 110] sample: 'To solve this problem, we need to understand the structure of the spiral pattern on the grid and identify the four numbers that lie on the same diagonal as the number \\(7\\).\n\n### Steps to Solve the Pr' +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06058003036542796, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 228}, "mem_gb": 16.05} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07366450755995077, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.3, "frames": {"chat": 221}, "mem_gb": 16.09} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05554351498146231, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.6, "frames": {"chat": 254}, "mem_gb": 15.89} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05564516919666591, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.8, "frames": {"chat": 210}, "mem_gb": 16.01} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.060165981625051546, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.24609375, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.3, "frames": {"chat": 226}, "mem_gb": 15.97} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06510323798443812, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 212}, "mem_gb": 16.04} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0749064519510294, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.7, "frames": {"chat": 211}, "mem_gb": 15.97} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07834269242317726, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.298828125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 196}, "mem_gb": 16.02} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.058797676620120184, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.9, "frames": {"chat": 212}, "mem_gb": 16.04} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.057397480336111036, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.1, "frames": {"chat": 244}, "mem_gb": 15.96} +[eval step 120] sample: 'To solve this problem, we need to understand the structure of the spiral pattern on the grid and identify the numbers that lie on the same diagonal as the number \\(7\\).\n\n### Steps to Solve the Problem' +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.061074561254587025, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.5, "frames": {"chat": 222}, "mem_gb": 16.0} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06682961131685103, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 218}, "mem_gb": 16.05} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0644194861006923, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.6, "frames": {"chat": 252}, "mem_gb": 15.92} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07198429885810861, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.4, "frames": {"chat": 202}, "mem_gb": 16.1} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07482979528958289, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.6, "frames": {"chat": 237}, "mem_gb": 16.05} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06636300194105134, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.7, "frames": {"chat": 234}, "mem_gb": 16.03} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0545189349170619, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 215}, "mem_gb": 16.05} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.053207050136492275, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.2431640625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.1, "frames": {"chat": 234}, "mem_gb": 15.98} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05247051075274746, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.2412109375, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 216}, "mem_gb": 16.03} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0620192121199177, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.5, "frames": {"chat": 210}, "mem_gb": 16.0} +[eval step 130] sample: 'To solve this problem, we need to identify the four numbers on the diagonal that include the number \\(7\\) and determine how many of these numbers are prime.\n\n### Steps to Solve the Problem:\n\n1. **Iden' +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06753606330246355, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.1, "frames": {"chat": 199}, "mem_gb": 16.05} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06196977580211436, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.9, "frames": {"chat": 210}, "mem_gb": 16.06} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05961425757134954, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 225}, "mem_gb": 16.0} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.058377123098867015, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.5, "frames": {"chat": 253}, "mem_gb": 15.9} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05910795236534128, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.9, "frames": {"chat": 247}, "mem_gb": 16.02} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06105499646520863, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.6, "frames": {"chat": 238}, "mem_gb": 16.02} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0698598912583664, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 235}, "mem_gb": 16.04} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0758493947354611, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.7, "frames": {"chat": 215}, "mem_gb": 16.02} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05895033346662919, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.8, "frames": {"chat": 268}, "mem_gb": 16.02} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06744063342887287, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.4, "frames": {"chat": 228}, "mem_gb": 16.05} +[eval step 140] sample: 'To solve this problem, we need to identify the four numbers on the diagonal that include the number \\(7\\) and determine how many of these numbers are prime.\n\n### Steps to Solve the Problem:\n\n1. **Iden' +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05960630811361286, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.5, "frames": {"chat": 252}, "mem_gb": 15.98} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.062250979524493835, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.7, "frames": {"chat": 223}, "mem_gb": 16.06} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07568410117311093, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.7, "frames": {"chat": 226}, "mem_gb": 16.05} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07676944995061494, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.0, "frames": {"chat": 208}, "mem_gb": 16.1} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05351945217698813, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.8, "frames": {"chat": 240}, "mem_gb": 15.98} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06589085518893165, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.0, "frames": {"chat": 222}, "mem_gb": 15.97} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05625805737182964, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.9, "frames": {"chat": 236}, "mem_gb": 16.05} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05725338245869304, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.9, "frames": {"chat": 217}, "mem_gb": 16.01} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.060780384954002994, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 252}, "mem_gb": 15.92} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05998445832872142, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 222}, "mem_gb": 16.04} +[eval step 150] sample: "To solve this problem, we need to identify the four numbers on the diagonal that are prime and lie on the same diagonal as the number \\(7\\). Let's break down the steps:\n\n1. **Identify the Diagonal Num" +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep50_s1225/step0150 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▅▅▂▇▁▄▅▆▄▆▇▁▂▆█▁▂▆▆▇▆▃▆▃▂▃▄▄▅▆▅▁▃▃▆▁▂▄▄▅ +wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▄▄▄▅▅▅▅▅▅▅▅▆▆▆▆▆▇▇▇▇█ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅███████████ +wandb: finish_rate ▂▂▆█▄▃▁█▄▅▅▆▅▆▆▃▃▃▂▅▇▃▄▆▃▄▄▄▅▄▄█▅▆▄▅█▅▅▅ +wandb: forward_topk_kl █▄▄▃▄▃▃▂▂▃▃▂▂▂▂▂▂▁▂▁▂▁▂▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm █▇▇▅▅▄▄▃▃▃▃▃▃▃▂▂▃▃▃▃▂▂▂▂▂▂▂▂▂▁▁▁▂▂▁▁▁▁▁▁ +wandb: lr ▁▃██████████████████████████████████████ +wandb: mem_gb ▅▅▅█▄▇▇▆▅▄▂▆▁▄▃▄▅▄▅▁▅▆▅▃▅▃▆▂▃▅▅▅▅▅▂▅▅▅▆▂ +wandb: step ▁▁▁▁▁▂▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▄▅▅▆▆▆▆▆▇▇▇▇▇▇███ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 540.5 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.847 +wandb: forward_topk_kl 0.05998 +wandb: grad_norm 0.26562 +wandb: lr 3e-05 +wandb: mem_gb 16.04 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run uniform-math-keep50-s1225 at: https://wandb.ai/hbfreed/glean-grid/runs/tj9y7tlh +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/uniform_keep50_s1225/wandb/run-20260716_000817-tj9y7tlh/logs +{ + "correct": 663, + "accuracy": 0.5026535253980288, + "finished": 1306, + "finish_rate": 0.9901440485216073, + "mean_completion_tokens": 111.62471569370736 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep50_s1225_step100_chat.json +{ + "correct": 676, + "accuracy": 0.5125094768764216, + "finished": 1307, + "finish_rate": 0.9909021986353298, + "mean_completion_tokens": 111.23199393479909 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep50_s1225_step150_chat.json diff --git a/healed/grid_math/uniform_keep75_s1224.console.log b/healed/grid_math/uniform_keep75_s1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..1f1e8954029ee9a44cc83bdf61c98a84729346f4 --- /dev/null +++ b/healed/grid_math/uniform_keep75_s1224.console.log @@ -0,0 +1,232 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run bknlxtxa +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/uniform_keep75_s1224/wandb/run-20260716_151320-bknlxtxa +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run uniform-math-keep75-s1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/bknlxtxa + Loading checkpoint shards: 0%| | 0/3 [00:00 outputs/healed/grid_math/uniform_keep75_s1224/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05692075071053114, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 222}, "mem_gb": 22.05} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.052173386886234706, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 235}, "mem_gb": 22.1} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.059462278134546555, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.3, "frames": {"chat": 208}, "mem_gb": 22.06} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.045240455083704244, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.4, "frames": {"chat": 191}, "mem_gb": 22.1} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03589170680805886, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.244140625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 219}, "mem_gb": 22.09} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03801956171578107, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 207}, "mem_gb": 22.1} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0399568536445188, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 208}, "mem_gb": 22.05} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03400595511964833, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.2177734375, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 219}, "mem_gb": 22.09} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03301748423215468, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.2216796875, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 246}, "mem_gb": 21.96} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.043141082558253156, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 206}, "mem_gb": 22.12} +[eval step 60] sample: "To solve the given system of equations with the constraint that each letter represents a non-zero digit, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\).\n\nLet's break down t" +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03528507920033298, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 233}, "mem_gb": 22.1} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03265478485104007, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.224609375, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 229}, "mem_gb": 21.96} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.031905491882663534, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.2255859375, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 236}, "mem_gb": 22.0} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03695723408527362, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.2373046875, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 239}, "mem_gb": 21.88} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.032592856184983005, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.2158203125, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 241}, "mem_gb": 22.0} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03586128164082766, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 226}, "mem_gb": 21.97} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03132054034647687, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.2275390625, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 234}, "mem_gb": 22.09} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.030788104644580743, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.224609375, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 257}, "mem_gb": 22.08} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.045981417168816555, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 208}, "mem_gb": 22.14} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04108849198470513, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 211}, "mem_gb": 22.11} +[eval step 70] sample: "To solve the given system of equations with the constraint that each letter represents a non-zero digit, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\).\n\nLet's break down t" +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03987854701854909, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 227}, "mem_gb": 22.1} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03947833718151475, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 211}, "mem_gb": 22.07} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.032092607042985034, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.23046875, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 238}, "mem_gb": 22.09} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03403336244607344, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.2216796875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 237}, "mem_gb": 22.13} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04188025526891773, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 208}, "mem_gb": 22.13} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.033074141753333, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.216796875, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 221}, "mem_gb": 22.22} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.032880204598419366, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 232}, "mem_gb": 22.05} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03631957043294484, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.23828125, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.5, "frames": {"chat": 208}, "mem_gb": 22.09} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.031250697735549574, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.2373046875, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 227}, "mem_gb": 22.01} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03714153531583336, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 221}, "mem_gb": 22.03} +[eval step 80] sample: "To solve the given system of equations with the constraint that each letter represents a non-zero digit, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\).\n\nLet's break down t" +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03134368731607683, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.2470703125, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 232}, "mem_gb": 22.1} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03455453577995456, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.2333984375, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.5, "frames": {"chat": 219}, "mem_gb": 22.1} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.036677736731572076, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.2275390625, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 195}, "mem_gb": 22.19} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03607522614639408, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.234375, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 216}, "mem_gb": 22.1} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.038589416616572995, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 208}, "mem_gb": 21.98} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03083962368282955, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.21875, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 235}, "mem_gb": 21.98} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03220725948303783, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 225}, "mem_gb": 22.08} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04238495488502085, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 213}, "mem_gb": 22.18} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.028955379137241593, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.2294921875, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 257}, "mem_gb": 21.85} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03898017764575779, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 212}, "mem_gb": 22.12} +[eval step 90] sample: "To solve the given system of equations with the constraint that each letter represents a non-zero digit, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\).\n\nLet's break down t" +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.033426403526705686, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.2451171875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 221}, "mem_gb": 22.1} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.030457712451570355, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.2109375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 242}, "mem_gb": 22.09} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03081075047897175, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.240234375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 220}, "mem_gb": 22.06} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02926128526229877, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.2265625, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 240}, "mem_gb": 21.95} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03344736107902912, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.228515625, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 206}, "mem_gb": 22.08} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03418054170239096, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.2294921875, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 226}, "mem_gb": 22.1} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03584913180077759, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 244}, "mem_gb": 21.88} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03905867359065451, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 224}, "mem_gb": 22.1} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.030537177202943713, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.2421875, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 271}, "mem_gb": 21.82} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03323106580647485, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 236}, "mem_gb": 22.11} +[eval step 100] sample: 'To solve the given system of equations with the constraint that each letter represents a non-zero digit, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\).\n\nThe equations are:' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep75_s1224/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.029700769117233964, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.234375, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 232}, "mem_gb": 21.97} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03277247970288154, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.2451171875, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 210}, "mem_gb": 22.03} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03219541934624625, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.2470703125, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 217}, "mem_gb": 22.0} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.035269686618377455, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.2451171875, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 224}, "mem_gb": 22.12} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.042453550912012965, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 203}, "mem_gb": 21.97} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03221358998257201, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.2255859375, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 239}, "mem_gb": 22.06} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0216342947023067, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.1875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 254}, "mem_gb": 21.98} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0223819046114649, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.2255859375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 241}, "mem_gb": 22.07} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0306436812187545, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 213}, "mem_gb": 22.1} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.027198022907366975, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 221}, "mem_gb": 22.15} +[eval step 110] sample: "To solve the given system of equations with the constraint that each letter represents a non-zero digit, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\).\n\nLet's break down t" +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.028086171217844822, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.3, "frames": {"chat": 196}, "mem_gb": 22.11} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021727316307936173, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.169921875, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 270}, "mem_gb": 21.91} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023765899055164, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.216796875, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 216}, "mem_gb": 22.09} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.027058855175063946, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 200}, "mem_gb": 22.06} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02359186636308829, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.1953125, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 206}, "mem_gb": 22.01} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.020846557306901863, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.1748046875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 234}, "mem_gb": 22.04} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02425564272745202, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.181640625, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 215}, "mem_gb": 22.05} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.020192126658698545, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.17578125, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 255}, "mem_gb": 22.04} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021318466197893335, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.177734375, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 250}, "mem_gb": 21.92} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022364306481989723, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.1728515625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 242}, "mem_gb": 22.09} +[eval step 120] sample: 'To solve the given system of equations with the constraint that each letter represents a non-zero digit, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) that satisfy all the equati' +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.027774021465998763, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.1953125, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 199}, "mem_gb": 22.1} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03040962266094672, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.216796875, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 208}, "mem_gb": 22.13} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02479745573766219, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.23046875, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.7, "frames": {"chat": 208}, "mem_gb": 22.07} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.029116933320866276, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.2099609375, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 209}, "mem_gb": 22.22} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.020989714213639186, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.17578125, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 235}, "mem_gb": 22.05} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022509423316735774, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.197265625, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 204}, "mem_gb": 22.04} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03046467731000545, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 208}, "mem_gb": 22.1} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022415347762413634, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 240}, "mem_gb": 22.09} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022870081384774917, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.189453125, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 246}, "mem_gb": 22.09} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02218571406165914, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.1796875, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 243}, "mem_gb": 21.91} +[eval step 130] sample: 'To solve the given system of equations with the constraint that each letter represents a non-zero digit, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\).\n\nThe equations are:' +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024033155613788403, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.1728515625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 208}, "mem_gb": 22.1} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.026436940440007797, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 219}, "mem_gb": 22.1} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03020671685578612, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 211}, "mem_gb": 22.11} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.025810199808344866, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.220703125, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 232}, "mem_gb": 22.07} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0242060411726047, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.1884765625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.3, "frames": {"chat": 214}, "mem_gb": 22.1} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023627944999715933, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.1748046875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 226}, "mem_gb": 21.99} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024266153483674863, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.3, "frames": {"chat": 210}, "mem_gb": 22.11} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023371724192472174, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 218}, "mem_gb": 21.93} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02139054679266798, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.1748046875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 233}, "mem_gb": 22.08} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02719521118995423, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.1982421875, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 215}, "mem_gb": 22.1} +[eval step 140] sample: 'To solve the given system of equations with the constraint that each letter represents a non-zero digit, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) that satisfy all the equati' +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02396822643155853, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.1865234375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 233}, "mem_gb": 22.08} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024230880417240162, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 209}, "mem_gb": 22.04} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01931162694086476, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.1689453125, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.1, "frames": {"chat": 262}, "mem_gb": 21.97} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022601187952638914, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.181640625, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 249}, "mem_gb": 22.06} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.026810239897008675, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.1875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.9, "frames": {"chat": 227}, "mem_gb": 22.09} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022271136725856924, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 221}, "mem_gb": 22.09} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02173946086432164, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.1640625, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 234}, "mem_gb": 22.1} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022538168500425913, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.1923828125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 213}, "mem_gb": 22.05} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022282669592516808, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.16796875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 213}, "mem_gb": 21.99} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.020306303256377577, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.1689453125, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 234}, "mem_gb": 22.02} +[eval step 150] sample: 'To solve the given system of equations with the constraint that each letter represents a non-zero digit, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) that satisfy all the equati' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep75_s1224/step0150 +wandb: uploading console lines 171-171; updating run metadata +wandb: uploading config.yaml; uploading output.log; uploading wandb-summary.json +wandb: +wandb: Run history: +wandb: comp_len ▅▇▆▄▃▄▃▅▅▆▅▆▃▇▄▃▅▆▄▅▆▅▄█▅▁▆▅▇▃▃▇▆▃▅▆▄▆▅▆ +wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▆▆▆▇▇▇▇▇▇███ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅██████████████ +wandb: finish_rate ▅▅▅█▇▆▂▃▃▇▄▄▅█▅█▃▅▆▅▁▃▃▁▆█▅▇▆▂▅▂▂▆▂▄▆█▇▅ +wandb: forward_topk_kl ▇█▅▅▅▄▃▄▃▃▃▄▃▃▃▃▂▂▂▂▂▂▂▂▂▁▂▂▂▂▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm ▇█▆▅▅▅▅▅▅▅▅▄▄▄▄▂▃▃▂▂▃▂▂▃▂▃▃▅▂▂▁▂▁▂▁▁▁▁▁▁ +wandb: lr ▁▅██████████████████████████████████████ +wandb: mem_gb ▆▃▅▄▃▆▆▅▃█▆▆▆▅▄▄▂▆▆▆▅▆▂▇▆▆▁▆▄▆▆▅▅▅▆▆▆▅▆▄ +wandb: step ▁▁▁▁▁▂▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▅▅▅▆▆▆▆▆▆▆▆▆▇▇▇▇█ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 512.8 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.906 +wandb: forward_topk_kl 0.02031 +wandb: grad_norm 0.16895 +wandb: lr 3e-05 +wandb: mem_gb 22.02 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run uniform-math-keep75-s1224 at: https://wandb.ai/hbfreed/glean-grid/runs/bknlxtxa +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/uniform_keep75_s1224/wandb/run-20260716_151320-bknlxtxa/logs +{ + "correct": 821, + "accuracy": 0.6224412433661866, + "finished": 1311, + "finish_rate": 0.9939347990902199, + "mean_completion_tokens": 117.59438968915845 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep75_s1224_step100_chat.json +{ + "correct": 836, + "accuracy": 0.6338134950720242, + "finished": 1313, + "finish_rate": 0.9954510993176648, + "mean_completion_tokens": 115.02274450341167 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep75_s1224_step150_chat.json diff --git a/healed/grid_math/uniform_keep75_s1225.console.log b/healed/grid_math/uniform_keep75_s1225.console.log new file mode 100644 index 0000000000000000000000000000000000000000..b98d4869bf7a50618c4c2d5cb4c3ab736c127283 --- /dev/null +++ b/healed/grid_math/uniform_keep75_s1225.console.log @@ -0,0 +1,232 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/uniform_keep75_s1225/wandb/run-20260716_223103-vjebg3kl +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run uniform-math-keep75-s1225 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/vjebg3kl + Loading checkpoint shards: 0%| | 0/3 [00:00 outputs/healed/grid_math/uniform_keep75_s1225/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06234172496208921, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 224}, "mem_gb": 22.12} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07160587412401413, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 203}, "mem_gb": 21.97} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05557245409963652, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 239}, "mem_gb": 22.06} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03414652680471384, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 254}, "mem_gb": 21.98} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.034510691293003035, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 241}, "mem_gb": 22.07} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04583843435803428, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 213}, "mem_gb": 22.1} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03947926825817364, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 221}, "mem_gb": 22.15} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04116059052028383, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 196}, "mem_gb": 22.11} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03308698924773683, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.224609375, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 270}, "mem_gb": 21.91} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.034069254077334576, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.2392578125, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 216}, "mem_gb": 22.09} +[eval step 60] sample: 'To solve this problem, we need to understand the structure of the spiral pattern on the grid and identify the numbers that lie on the same diagonal as the number \\(7\\).\n\n### Step-by-Step Solution:\n\n1.' +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04350406211614609, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 200}, "mem_gb": 22.06} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.033727901365452756, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.2412109375, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 206}, "mem_gb": 22.01} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03015216571188066, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.2314453125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 234}, "mem_gb": 22.04} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03634754524397819, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.24609375, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 215}, "mem_gb": 22.05} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.029129735330779415, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.2314453125, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 255}, "mem_gb": 22.04} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03348110796995461, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 250}, "mem_gb": 21.92} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03267918671743634, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 242}, "mem_gb": 22.09} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0429476497222359, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 199}, "mem_gb": 22.1} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04425143690695986, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 208}, "mem_gb": 22.13} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03657152016861364, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 208}, "mem_gb": 22.07} +[eval step 70] sample: 'To solve this problem, we need to understand the structure of the spiral pattern on the square grid and identify the numbers that lie on the same diagonal as the number \\(7\\).\n\n### Step-by-Step Soluti' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04123211079263128, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 209}, "mem_gb": 22.22} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.030864056057242364, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.244140625, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 235}, "mem_gb": 22.05} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.033988771005704375, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 204}, "mem_gb": 22.04} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.043746256581383446, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.298828125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 208}, "mem_gb": 22.1} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.031305211680731734, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.228515625, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.3, "frames": {"chat": 240}, "mem_gb": 22.09} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03433587677682129, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.8, "frames": {"chat": 246}, "mem_gb": 22.09} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03481989274016426, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.6, "frames": {"chat": 243}, "mem_gb": 21.91} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.036875585224215565, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.0, "frames": {"chat": 208}, "mem_gb": 22.1} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03534174345984745, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.2392578125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.8, "frames": {"chat": 219}, "mem_gb": 22.1} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.042089351415184016, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.9, "frames": {"chat": 211}, "mem_gb": 22.11} +[eval step 80] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid and identify the four shaded squares that lie on the same diagonal as the number 7. Then, we wil' +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03182697140184076, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.23828125, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.1, "frames": {"chat": 232}, "mem_gb": 22.07} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.034447449180250986, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.2392578125, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.7, "frames": {"chat": 214}, "mem_gb": 22.1} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03462826530029997, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.6, "frames": {"chat": 226}, "mem_gb": 21.99} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0355406848211928, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.2490234375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.1, "frames": {"chat": 210}, "mem_gb": 22.11} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03192530898133603, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 218}, "mem_gb": 21.93} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.030032073534412, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.2080078125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.0, "frames": {"chat": 233}, "mem_gb": 22.08} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0384073818150054, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.2412109375, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.2, "frames": {"chat": 215}, "mem_gb": 22.1} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03470420025307685, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 233}, "mem_gb": 22.08} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03370989526286721, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 209}, "mem_gb": 22.04} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.027565786793866814, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.197265625, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.9, "frames": {"chat": 262}, "mem_gb": 21.97} +[eval step 90] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid and identify the four shaded squares that lie on the same diagonal as the number 7. Then, we wil' +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03008150718464361, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.21875, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.6, "frames": {"chat": 249}, "mem_gb": 22.06} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03847663177931681, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.6, "frames": {"chat": 227}, "mem_gb": 22.09} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.032657393121114, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.1, "frames": {"chat": 221}, "mem_gb": 22.09} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03173800096526587, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.2265625, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.1, "frames": {"chat": 234}, "mem_gb": 22.1} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03087014857486356, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.2470703125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.7, "frames": {"chat": 213}, "mem_gb": 22.05} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.029955804504655924, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.2236328125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 213}, "mem_gb": 21.99} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.030467382055746083, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.24609375, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 55.1, "frames": {"chat": 234}, "mem_gb": 22.02} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03673412882296058, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.4, "frames": {"chat": 222}, "mem_gb": 22.09} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.033574912037172666, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.23046875, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.6, "frames": {"chat": 227}, "mem_gb": 22.1} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.034915446859543835, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.7, "frames": {"chat": 218}, "mem_gb": 22.14} +[eval step 100] sample: 'To solve this problem, we need to understand the structure of the spiral pattern formed by the numbers from 1 to 49 on a square grid. The numbers are arranged in a spiral pattern, starting from the ce' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep75_s1225/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.033782800041108084, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.21875, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.7, "frames": {"chat": 223}, "mem_gb": 22.1} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03598464701743796, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.24609375, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.2, "frames": {"chat": 206}, "mem_gb": 22.1} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.030618002940326308, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.2412109375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 213}, "mem_gb": 22.02} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04078237585676834, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 223}, "mem_gb": 21.96} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03272634071402718, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.2197265625, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 227}, "mem_gb": 22.07} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03005558260572143, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.2421875, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 253}, "mem_gb": 22.09} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.026308241902609976, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 216}, "mem_gb": 22.1} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02579004672653197, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.201171875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 205}, "mem_gb": 22.07} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.030623514922133957, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.20703125, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 207}, "mem_gb": 22.16} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02862321507255547, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.2001953125, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 215}, "mem_gb": 22.08} +[eval step 110] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid starting from the center. We then identify the four shaded squares that lie on the same diagonal' +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023055778127916468, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 228}, "mem_gb": 22.1} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.025379221981678468, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.189453125, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 221}, "mem_gb": 22.14} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.019185258447751402, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.1748046875, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 254}, "mem_gb": 21.93} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02264149820668778, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.2275390625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 210}, "mem_gb": 22.06} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021573141938253926, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.1650390625, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 226}, "mem_gb": 22.02} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024346172539202963, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 212}, "mem_gb": 22.09} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02802123403872053, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.2060546875, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 211}, "mem_gb": 22.02} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.026772187854287526, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.1, "frames": {"chat": 196}, "mem_gb": 22.07} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02290341926627637, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.1884765625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.3, "frames": {"chat": 212}, "mem_gb": 22.09} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02147745310171352, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.1796875, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 244}, "mem_gb": 22.0} +[eval step 120] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid starting from the center. We then identify the four shaded squares that lie on the same diagonal' +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02290404658280313, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.173828125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 222}, "mem_gb": 22.05} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0237099109623814, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.2021484375, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.7, "frames": {"chat": 218}, "mem_gb": 22.09} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02240949104845058, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.1708984375, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 252}, "mem_gb": 21.97} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02601201236327179, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.1904296875, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 202}, "mem_gb": 22.14} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0266425287119889, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.1923828125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 237}, "mem_gb": 22.1} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02409238005452013, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.17578125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 234}, "mem_gb": 22.08} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021687637067609466, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.201171875, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 215}, "mem_gb": 22.1} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.019817282370629255, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.1669921875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 234}, "mem_gb": 22.03} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.020775739161809907, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.1708984375, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.2, "frames": {"chat": 216}, "mem_gb": 22.08} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023808946748528008, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.1865234375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.5, "frames": {"chat": 210}, "mem_gb": 22.05} +[eval step 130] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid and identify the four shaded squares that lie on the same diagonal as the number 7. Then, we wil' +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02468363624312915, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.189453125, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.3, "frames": {"chat": 199}, "mem_gb": 22.1} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023548281814196766, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.3, "frames": {"chat": 210}, "mem_gb": 22.11} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021664702919110036, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.173828125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 225}, "mem_gb": 22.05} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021577526232468273, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.173828125, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 253}, "mem_gb": 21.95} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02105618586166917, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.166015625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 247}, "mem_gb": 22.07} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02262616546676339, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.177734375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 238}, "mem_gb": 22.07} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024266567853023297, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.1787109375, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 235}, "mem_gb": 22.09} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.025859414496971295, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.2001953125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 215}, "mem_gb": 22.06} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021803223936345116, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 268}, "mem_gb": 22.06} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024701077461933407, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.1953125, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 228}, "mem_gb": 22.1} +[eval step 140] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid and identify the four shaded squares that lie on the same diagonal as the number 7. Then, we wil' +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.020695444104556614, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.162109375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 252}, "mem_gb": 22.03} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023492466158481936, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.2, "frames": {"chat": 223}, "mem_gb": 22.11} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.026726572511414998, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 226}, "mem_gb": 22.09} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02891799058924274, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 208}, "mem_gb": 22.14} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02066739065583485, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.1865234375, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 240}, "mem_gb": 22.03} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023900935378763824, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.1806640625, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 222}, "mem_gb": 22.02} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02020211691574271, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.1552734375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 236}, "mem_gb": 22.09} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022802902391382184, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.1923828125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.4, "frames": {"chat": 217}, "mem_gb": 22.06} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02172313815508193, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.1923828125, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 252}, "mem_gb": 21.97} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022857091062360755, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.1875, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 222}, "mem_gb": 22.08} +[eval step 150] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid and identify the four shaded squares that lie on the same diagonal as the number 7. We then need' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep75_s1225/step0150 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▇▅▄▇▄▄▆▆▅▃▄▃█▁▂▆▃▃▆▄▅▄▂▃▅▆▇▆▅▆█▅▇▆▃▅▃▅▄▅ +wandb: cumulative_loss_tokens ▁▁▁▂▂▂▂▃▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅██████████ +wandb: finish_rate ▆▆▇█▁▆▅▄▅█▅▇▁▅▃▆▃▇█▇▇▇▄▅▃▆▅▄▄▃▅▆▁▅▅█▅▇▁▇ +wandb: forward_topk_kl ▇█▆▅▆▆▅▅▇▃▃▂▂▃▂▄▃▃▃▃▂▂▂▃▃▁▂▁▁▁▂▁▁▁▂▁▁▂▂▁ +wandb: grad_norm █▄▃▂▂▂▂▂▂▂▂▂▂▂▂▂▂▁▁▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁███████████████████████████████████████ +wandb: mem_gb ▆▃▇▅▇▄▃█▅▆▃▆▆▁▆▇▅▅▄▁▆▅▆▅▆▄▆▆▅▇▄▆▇▅▆▅▅▇▄▅ +wandb: step ▁▁▁▂▂▂▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▅▅▅▅▅▅▅▅▆▆▆▆▆▇▇▇▇██ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 540.5 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.847 +wandb: forward_topk_kl 0.02286 +wandb: grad_norm 0.1875 +wandb: lr 3e-05 +wandb: mem_gb 22.08 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run uniform-math-keep75-s1225 at: https://wandb.ai/hbfreed/glean-grid/runs/vjebg3kl +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/uniform_keep75_s1225/wandb/run-20260716_223103-vjebg3kl/logs +{ + "correct": 840, + "accuracy": 0.6368460955269143, + "finished": 1311, + "finish_rate": 0.9939347990902199, + "mean_completion_tokens": 116.58377558756634 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep75_s1225_step100_chat.json +{ + "correct": 845, + "accuracy": 0.640636846095527, + "finished": 1312, + "finish_rate": 0.9946929492039424, + "mean_completion_tokens": 113.11978771796815 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep75_s1225_step150_chat.json diff --git a/healed/grid_math/uniform_keep75_s1226.console.log b/healed/grid_math/uniform_keep75_s1226.console.log new file mode 100644 index 0000000000000000000000000000000000000000..7b3971d5401b1e6fe9ab5aff27064fb8fd4d7d7e --- /dev/null +++ b/healed/grid_math/uniform_keep75_s1226.console.log @@ -0,0 +1,248 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/grid_math/uniform_keep75_s1226/wandb/run-20260716_194207-t8hs7m3b +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run uniform-math-keep75-s1226 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/t8hs7m3b + Loading checkpoint shards: 0%| | 0/3 [00:00 outputs/healed/grid_math/uniform_keep75_s1226/step0050 +{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07032583207286273, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 223}, "mem_gb": 21.96} +{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.059393302978233746, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 227}, "mem_gb": 22.07} +{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05184000949463807, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 253}, "mem_gb": 22.09} +{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.038828376208990815, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 216}, "mem_gb": 22.1} +{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.039238105912537624, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 205}, "mem_gb": 22.07} +{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.047572107557269434, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 207}, "mem_gb": 22.16} +{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04661659479457885, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 215}, "mem_gb": 22.08} +{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.034201694558095186, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 228}, "mem_gb": 22.1} +{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04092416759391005, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 221}, "mem_gb": 22.14} +{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.029803095571851977, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 254}, "mem_gb": 21.93} +[eval step 60] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of the sides of a triangle. This new triangle is known as the medial triangle, ' +{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03530521812000467, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 210}, "mem_gb": 22.06} +{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.033094389010050025, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.21484375, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 226}, "mem_gb": 22.02} +{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03637227460097832, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 212}, "mem_gb": 22.09} +{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.041987014663917945, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 211}, "mem_gb": 22.02} +{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03841204237483131, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.3, "frames": {"chat": 196}, "mem_gb": 22.07} +{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03594996488027585, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 212}, "mem_gb": 22.09} +{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.032274080974639706, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.2177734375, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 244}, "mem_gb": 22.0} +{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03448318504060929, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.2392578125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 222}, "mem_gb": 22.05} +{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.035316669923362014, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 218}, "mem_gb": 22.09} +{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0348655257951313, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.2353515625, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 252}, "mem_gb": 21.97} +[eval step 70] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of the sides of a given triangle.\n\n### Steps to Solve:\n\n1. **Understand the Pro' +{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03988810336939059, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 202}, "mem_gb": 22.14} +{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04227977843079716, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 237}, "mem_gb": 22.1} +{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03656210743497747, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.234375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 234}, "mem_gb": 22.08} +{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03134735133058081, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 215}, "mem_gb": 22.1} +{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03046899615646495, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.216796875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 234}, "mem_gb": 22.03} +{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.029838484331507546, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.2099609375, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 216}, "mem_gb": 22.08} +{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03392047594260269, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.248046875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 210}, "mem_gb": 22.05} +{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03507764483672411, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 199}, "mem_gb": 22.1} +{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03239439557224202, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.2216796875, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 210}, "mem_gb": 22.11} +{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03286070772272845, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 225}, "mem_gb": 22.05} +[eval step 80] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of the sides of a given triangle.\n\n### Steps to Solve:\n\n1. **Understand the Mid' +{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.032128432613688834, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 253}, "mem_gb": 21.95} +{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.031100997417598652, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.21875, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 247}, "mem_gb": 22.07} +{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0324900306135416, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.2412109375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 238}, "mem_gb": 22.07} +{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03485131102278829, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.2451171875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 235}, "mem_gb": 22.09} +{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03912405168330297, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 215}, "mem_gb": 22.06} +{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.033098834516915185, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 268}, "mem_gb": 22.06} +{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.034645479888437934, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 228}, "mem_gb": 22.1} +{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03237436599015103, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.240234375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 252}, "mem_gb": 22.03} +{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03435483168452047, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 223}, "mem_gb": 22.11} +{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04248241032837735, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 226}, "mem_gb": 22.09} +[eval step 90] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of the sides of a triangle.\n\n### Steps to Solve the Problem:\n\n1. **Understand t' +{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04142716025705449, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 208}, "mem_gb": 22.14} +{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.031166394373402, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 240}, "mem_gb": 22.03} +{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03443516800537861, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.23046875, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 222}, "mem_gb": 22.02} +{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.030224586171346407, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.2333984375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 236}, "mem_gb": 22.09} +{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03236451368537576, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.2353515625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 217}, "mem_gb": 22.06} +{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.034898156173845445, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 252}, "mem_gb": 21.97} +{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02913232143558562, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.2255859375, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 222}, "mem_gb": 22.08} +{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.029040355308757475, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.21484375, "lr": 3e-05, "finish_rate": 0.901, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 242}, "mem_gb": 21.97} +{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.041233757554925976, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 219}, "mem_gb": 22.03} +{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.031569313709802614, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 223}, "mem_gb": 22.04} +[eval step 100] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of the sides of a given triangle.\n\n### Steps to Solve:\n\n1. **Understand the Pro' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep75_s1226/step0100 +{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03175070390204589, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.228515625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 232}, "mem_gb": 22.04} +{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03924315899686578, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.832, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 220}, "mem_gb": 22.09} +{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03732912681673964, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.236328125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 210}, "mem_gb": 22.14} +{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03158854953125119, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 226}, "mem_gb": 22.06} +{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.033342002084447694, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.741, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 212}, "mem_gb": 22.09} +{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.030606075126398354, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 236}, "mem_gb": 22.1} +{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021876435576969135, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.2001953125, "lr": 3e-05, "finish_rate": 0.928, "comp_len": 454.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 264}, "mem_gb": 21.97} +{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.031198739765095525, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.21875, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 229}, "mem_gb": 22.07} +{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02165067300312221, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.1806640625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 465.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 258}, "mem_gb": 21.95} +{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.028640677657909692, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 208}, "mem_gb": 22.11} +[eval step 110] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of the sides of a given triangle.\n\n### Steps to Solve:\n\n1. **Understand the Pro' +{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022872452891978902, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.88, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 249}, "mem_gb": 22.02} +{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02100934052404482, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.19140625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 220}, "mem_gb": 22.09} +{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024147335090193275, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.1806640625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 223}, "mem_gb": 22.08} +{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024052943600167055, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.201171875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.7, "frames": {"chat": 221}, "mem_gb": 22.09} +{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02166470361133882, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.162109375, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 230}, "mem_gb": 22.0} +{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02461200471681077, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.171875, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 214}, "mem_gb": 22.07} +{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03192179344429945, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.224609375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 214}, "mem_gb": 22.08} +{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.027875469126327275, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 210}, "mem_gb": 22.13} +{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.028064899455732668, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 214}, "mem_gb": 22.09} +{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02598108633084533, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.791, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 215}, "mem_gb": 22.05} +[eval step 120] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of the sides of a given triangle.\n\n### Steps to Solve:\n\n1. **Understand the Pro' +{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.029542491081776097, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.2060546875, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 208}, "mem_gb": 22.09} +{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023529452410464485, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.201171875, "lr": 3e-05, "finish_rate": 0.789, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 218}, "mem_gb": 21.97} +{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021247040836008577, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.16015625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 233}, "mem_gb": 21.99} +{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02059493543021381, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.173828125, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 231}, "mem_gb": 22.03} +{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024373427555907982, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 235}, "mem_gb": 22.22} +{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02307887886830916, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.1650390625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 216}, "mem_gb": 22.08} +{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02126138604179335, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.162109375, "lr": 3e-05, "finish_rate": 0.831, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 237}, "mem_gb": 22.1} +{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.029153408732265233, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.2001953125, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 203}, "mem_gb": 22.11} +{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02260789417422687, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.17578125, "lr": 3e-05, "finish_rate": 0.873, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 236}, "mem_gb": 22.13} +{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022835769553498055, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.1728515625, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 214}, "mem_gb": 22.1} +[eval step 130] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of the sides of a triangle. This new triangle is known as the medial triangle, ' +{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.026547468842783323, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.189453125, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.3, "frames": {"chat": 209}, "mem_gb": 22.09} +{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022498934920489166, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.1826171875, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 256}, "mem_gb": 22.1} +{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021946324599829193, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.1787109375, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 230}, "mem_gb": 21.96} +{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024039489145313078, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 230}, "mem_gb": 22.15} +{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02589866460277699, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.1796875, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 227}, "mem_gb": 22.05} +{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023664263719754913, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.166015625, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 208}, "mem_gb": 22.11} +{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.027114476467870796, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.1796875, "lr": 3e-05, "finish_rate": 0.699, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 206}, "mem_gb": 22.12} +{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02399630657126351, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.1767578125, "lr": 3e-05, "finish_rate": 0.82, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 228}, "mem_gb": 22.0} +{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02343961157395194, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.181640625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 224}, "mem_gb": 22.09} +{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024890781568270178, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.66, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 200}, "mem_gb": 22.13} +[eval step 140] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of the sides of a given triangle.\n\n### Steps to Solve the Problem:\n\n1. **Unders' +{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02478467862996428, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.1904296875, "lr": 3e-05, "finish_rate": 0.714, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 196}, "mem_gb": 22.11} +{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022272795487005108, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.169921875, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 223}, "mem_gb": 22.09} +{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022915525561606045, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.869, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 213}, "mem_gb": 21.98} +{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0212597823954653, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.1953125, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 232}, "mem_gb": 22.02} +{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.020988663152477237, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.1796875, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 223}, "mem_gb": 22.02} +{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021486785067555806, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.1630859375, "lr": 3e-05, "finish_rate": 0.85, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 233}, "mem_gb": 22.11} +{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02550031017503546, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.2021484375, "lr": 3e-05, "finish_rate": 0.816, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 217}, "mem_gb": 22.11} +{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.030582861898102175, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.752, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 202}, "mem_gb": 22.17} +{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021075663964031262, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.1640625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.1, "frames": {"chat": 253}, "mem_gb": 22.03} +{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02149440993195555, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 231}, "mem_gb": 22.03} +[eval step 150] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of the sides of a given triangle.\n\n### Steps to Solve:\n\n1. **Understand the Pro' +checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep75_s1226/step0150 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▁▅▇▇▇▅▇▆▄▆▃▄▇▆▆▄▄▇▆▁▂▄▄▇▃▆▂▄▃▂▆▆▇▆▄▃█▇▆▄ +wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▂▃▃▃▃▃▃▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇████ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅██████████████ +wandb: finish_rate ▇▆▂█▄▇▄▃▄▃▇▆▄▃▃▇▄▁▄▆▄▆▁▇▄▇▅█▆▅▅▁▃▆▆▇▅▅▁▂ +wandb: forward_topk_kl ▇█▅▄▄▃▃▃▃▃▃▂▂▂▂▂▂▂▂▂▂▂▂▂▂▂▁▂▂▂▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm █▄▃▃▃▃▂▂▂▂▂▂▂▂▁▂▁▁▁▂▁▂▁▂▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▂ +wandb: lr ▁███████████████████████████████████████ +wandb: mem_gb ▅▆▅▄▁▅█▅▅▅▅▅▅▆▅▅▆▁▄▅▅▄▂▅▅▆▃▂▄▅▃▅▃▅█▂▆▅▃▇ +wandb: step ▁▁▁▁▁▂▂▂▂▂▂▂▃▃▃▃▃▃▃▃▃▄▄▄▄▅▅▅▆▆▇▇▇▇▇▇████ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 519.5 +wandb: cumulative_loss_tokens 18000000 +wandb: epoch 2 +wandb: finish_rate 0.879 +wandb: forward_topk_kl 0.02149 +wandb: grad_norm 0.30859 +wandb: lr 3e-05 +wandb: mem_gb 22.03 +wandb: step 150 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run uniform-math-keep75-s1226 at: https://wandb.ai/hbfreed/glean-grid/runs/ta2qddpy +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/grid_math/uniform_keep75_s1226/wandb/run-20260716_223215-ta2qddpy/logs +{ + "correct": 826, + "accuracy": 0.6262319939347991, + "finished": 1314, + "finish_rate": 0.9962092494313874, + "mean_completion_tokens": 114.73464746019712 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep75_s1226_step100_chat.json +{ + "correct": 834, + "accuracy": 0.6322971948445792, + "finished": 1317, + "finish_rate": 0.9984836997725549, + "mean_completion_tokens": 114.99090219863533 +} +saved item-level results -> outputs/evals/grid_math/uniform_keep75_s1226_step150_chat.json diff --git a/healed/grid_math/worker_s1226.log b/healed/grid_math/worker_s1226.log new file mode 100644 index 0000000000000000000000000000000000000000..d63b8b0df1f407a9b9a18ba003b1288d03111d5a --- /dev/null +++ b/healed/grid_math/worker_s1226.log @@ -0,0 +1,22 @@ +2026-07-16T00:08:10-07:00 [s1226] glean_keep50_s1226 already done, skip +2026-07-16T00:08:10-07:00 [s1226] healing uniform_keep50_s1226 on GPU-864c54df +2026-07-16T01:41:59-07:00 [s1226] eval uniform_keep50_s1226 step100 +2026-07-16T01:44:01-07:00 [s1226] eval uniform_keep50_s1226 step150 +2026-07-16T01:45:55-07:00 [s1226] uniform_keep50_s1226 done -> 0.5049279757391963 +2026-07-16T01:45:55-07:00 [s1226] healing reap_keep50_s1226 on GPU-864c54df +2026-07-16T03:44:03-07:00 [s1226] eval reap_keep50_s1226 step100 +2026-07-16T03:45:49-07:00 [s1226] eval reap_keep50_s1226 step150 +2026-07-16T03:47:29-07:00 [s1226] reap_keep50_s1226 done -> 0.5913570887035633 +2026-07-16T03:47:29-07:00 [s1226] healing glean_keep25_s1226 on GPU-864c54df +2026-07-16T05:11:36-07:00 [s1226] eval glean_keep25_s1226 step100 +2026-07-16T05:13:54-07:00 [s1226] eval glean_keep25_s1226 step150 +2026-07-16T05:16:13-07:00 [s1226] glean_keep25_s1226 done -> 0.4359363153904473 +2026-07-16T05:16:13-07:00 [s1226] healing uniform_keep25_s1226 on GPU-864c54df +2026-07-16T06:29:43-07:00 [s1226] eval uniform_keep25_s1226 step100 +2026-07-16T06:31:48-07:00 [s1226] eval uniform_keep25_s1226 step150 +2026-07-16T06:33:45-07:00 [s1226] uniform_keep25_s1226 done -> 0.22744503411675512 +2026-07-16T06:33:45-07:00 [s1226] healing reap_keep25_s1226 on GPU-864c54df +2026-07-16T08:23:55-07:00 [s1226] eval reap_keep25_s1226 step100 +2026-07-16T08:26:19-07:00 [s1226] eval reap_keep25_s1226 step150 +2026-07-16T08:28:37-07:00 [s1226] reap_keep25_s1226 done -> 0.12357846853677028 +2026-07-16T08:28:38-07:00 [s1226] healing glean_keep75_s1226 on GPU-864c54df diff --git a/healed/liger_mb8/args.json b/healed/liger_mb8/args.json new file mode 100644 index 0000000000000000000000000000000000000000..67dc86adeb6871b419d1752adce0ce868c4fad15 --- /dev/null +++ b/healed/liger_mb8/args.json @@ -0,0 +1,70 @@ +{ + "student": "outputs/pruned/glean-0125inst-math-keep50", + "teacher": "allenai/OLMoE-1B-7B-0125-Instruct", + "training_mode": "on-policy", + "kl_direction": "reverse", + "dataset": "allenai/Dolci-Instruct-RL", + "dataset_sources": null, + "max_difficulty": null, + "trajectories": "outputs/teacher_trajectories/dolci_math_curated.jsonl", + "trajectory_dataset": "allenai/Dolci-Instruct-RL", + "off_policy_frames": "chat", + "off_policy_max_seq_len": 2048, + "topk_targets": null, + "max_loss_tokens": null, + "loss_tokens_per_step": null, + "teacher_device": "cuda:0", + "student_device": "cuda:1", + "lr": 3e-05, + "optimizer": "adamw8bit", + "weight_decay": 0.1, + "epochs": 1, + "prompts_per_step": 64, + "group_size": 4, + "rollout_batch": 64, + "micro_batch": 8, + "max_new_tokens": 2048, + "max_prompt_len": 1024, + "warmup_steps": 10, + "max_grad_norm": 1.0, + "eval_every": 10, + "gsm8k_every": 0, + "gsm8k_n": 256, + "gsm8k_batch": 16, + "gsm8k_max_new_tokens": 512, + "gsm8k_frames": "chat", + "save_every": 1000, + "out_dir": "outputs/healed/liger_mb8", + "sweep": 4, + "wandb": false, + "wandb_project": "glean-heal", + "wandb_run_name": null, + "wandb_run_id": null, + "wandb_resume": null, + "wandb_mode": "offline", + "no_wandb_sync": true, + "debug": false, + "resume_from": null, + "start_step": 0, + "no_grad_checkpointing": false, + "seed": 1223, + "no_teacher_overlap": false, + "sync_checkpoints": false, + "rollout_engine": "vllm", + "vllm_gpu": "2", + "vllm_port": 8377, + "vllm_refresh_every": 1, + "vllm_serve_bin": "vllm-plugin/.venv/bin/python", + "vllm_gpu_mem_util": 0.85, + "liger_loss": true, + "gold_mix_lambda": 0.5, + "gold_topk_targets": "outputs/teacher_trajectories/dolci_combined_top128", + "gold_mix_decay": 0.0, + "fast_teacher": true, + "reference_kl_beta": 0.05, + "drop_truncated_rollouts": false, + "vllm_max_model_len": null, + "vllm_refresh_mode": "reload", + "vllm_live_dir": null, + "resolved_kl_direction": "reverse" +} \ No newline at end of file diff --git a/healed/liger_mb8/train_log.jsonl b/healed/liger_mb8/train_log.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..9655c22e5ff1cae4d02032d3c6bec5abca56688b --- /dev/null +++ b/healed/liger_mb8/train_log.jsonl @@ -0,0 +1,4 @@ +{"step": 1, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6749841483754777, "tokens": 39094, "cumulative_loss_tokens": 39094, "grad_norm": 5.34375, "lr": 6e-06, "finish_rate": 1.0, "comp_len": 610.8, "dropped_truncated": 0, "gold_loss": 0.2627, "gold_lambda": 0.5, "rep_ratio": 2.53, "t_data_s": 0.0, "t_rollout_s": 25.4, "t_step_s": 70.5, "t_refresh_s": 0.3, "mem_gb": 12.14, "mem_gb_teacher": 20.77} +{"step": 2, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7909456714061366, "tokens": 36996, "cumulative_loss_tokens": 76090, "grad_norm": 6.625, "lr": 9e-06, "finish_rate": 0.984, "comp_len": 578.1, "dropped_truncated": 0, "gold_loss": 0.3092, "gold_lambda": 0.5, "rep_ratio": 2.287, "t_data_s": 0.0, "t_rollout_s": 29.7, "t_step_s": 57.0, "t_refresh_s": 0.3, "mem_gb": 11.99, "mem_gb_teacher": 21.07} +{"step": 3, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7638091775618511, "tokens": 50596, "cumulative_loss_tokens": 126686, "grad_norm": 5.5, "lr": 1.2e-05, "finish_rate": 1.0, "comp_len": 790.6, "dropped_truncated": 0, "gold_loss": 0.2744, "gold_lambda": 0.5, "rep_ratio": 2.382, "t_data_s": 0.0, "t_rollout_s": 29.5, "t_step_s": 59.7, "t_refresh_s": 0.3, "mem_gb": 12.08, "mem_gb_teacher": 20.96} +{"step": 4, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3501273521456831, "tokens": 46928, "cumulative_loss_tokens": 173614, "grad_norm": 1.8828125, "lr": 1.5e-05, "finish_rate": 0.984, "comp_len": 733.2, "dropped_truncated": 0, "gold_loss": 0.2477, "gold_lambda": 0.5, "rep_ratio": 2.585, "t_data_s": 0.0, "t_rollout_s": 30.6, "t_step_s": 49.4, "t_refresh_s": 0.0, "mem_gb": 12.07, "mem_gb_teacher": 21.07} diff --git a/healed/liger_mb8/vllm_server.log b/healed/liger_mb8/vllm_server.log new file mode 100644 index 0000000000000000000000000000000000000000..6f4704b5c6539d8f38eda574bf56149cdf55e1db --- /dev/null +++ b/healed/liger_mb8/vllm_server.log @@ -0,0 +1,309 @@ +Skipping import of cpp extensions due to incompatible torch version 2.10.0+cu128 for torchao version 0.15.0 Please see https://github.com/pytorch/ao/issues/2919 for more info +WARNING 07-30 21:52:23 [registry.py:915] Model architecture OlmoeForCausalLM is already registered, and will be overwritten by the new model class glean_vllm.pruned_olmoe:PrunedOlmoeForCausalLM. +(APIServer pid=2023926) INFO 07-30 21:52:23 [utils.py:299] +(APIServer pid=2023926) INFO 07-30 21:52:23 [utils.py:299] █ █ █▄ ▄█ +(APIServer pid=2023926) INFO 07-30 21:52:23 [utils.py:299] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.19.0 +(APIServer pid=2023926) INFO 07-30 21:52:23 [utils.py:299] █▄█▀ █ █ █ █ model outputs/pruned/glean-0125inst-math-keep50 +(APIServer pid=2023926) INFO 07-30 21:52:23 [utils.py:299] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀ +(APIServer pid=2023926) INFO 07-30 21:52:23 [utils.py:299] +(APIServer pid=2023926) INFO 07-30 21:52:23 [utils.py:233] non-default args: {'model_tag': 'outputs/pruned/glean-0125inst-math-keep50', 'host': '127.0.0.1', 'port': 8377, 'model': 'outputs/pruned/glean-0125inst-math-keep50', 'max_model_len': 3200, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85} +(APIServer pid=2023926) INFO 07-30 21:52:31 [model.py:549] Resolved architecture: OlmoeForCausalLM +(APIServer pid=2023926) INFO 07-30 21:52:31 [model.py:1678] Using max model len 3200 +(APIServer pid=2023926) INFO 07-30 21:52:31 [vllm.py:790] Asynchronous scheduling is enabled. +(APIServer pid=2023926) WARNING 07-30 21:52:31 [vllm.py:848] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none +(APIServer pid=2023926) WARNING 07-30 21:52:31 [vllm.py:859] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored. +(APIServer pid=2023926) INFO 07-30 21:52:31 [vllm.py:1025] Cudagraph is disabled under eager mode +(APIServer pid=2023926) INFO 07-30 21:52:31 [compilation.py:290] Enabled custom fusions: norm_quant, act_quant +Skipping import of cpp extensions due to incompatible torch version 2.10.0+cu128 for torchao version 0.15.0 Please see https://github.com/pytorch/ao/issues/2919 for more info +(EngineCore pid=2024236) WARNING 07-30 21:52:39 [registry.py:915] Model architecture OlmoeForCausalLM is already registered, and will be overwritten by the new model class glean_vllm.pruned_olmoe:PrunedOlmoeForCausalLM. +(EngineCore pid=2024236) INFO 07-30 21:52:39 [core.py:105] Initializing a V1 LLM engine (v0.19.0) with config: model='outputs/pruned/glean-0125inst-math-keep50', speculative_config=None, tokenizer='outputs/pruned/glean-0125inst-math-keep50', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=3200, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=2024236) INFO 07-30 21:52:39 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:35469 backend=nccl +(EngineCore pid=2024236) INFO 07-30 21:52:39 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=2024236) INFO 07-30 21:52:40 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep50... +(EngineCore pid=2024236) INFO 07-30 21:52:40 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=2024236) INFO 07-30 21:52:40 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=2024236) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=2018864) INFO 07-30 20:41:53 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:40655 backend=nccl +(EngineCore pid=2018864) INFO 07-30 20:41:53 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=2018864) INFO 07-30 20:41:54 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep50... +(EngineCore pid=2018864) INFO 07-30 20:41:55 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=2018864) INFO 07-30 20:41:55 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=2018864) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=11061) INFO 08-01 12:01:01 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.27:52855 backend=nccl +(EngineCore pid=11061) INFO 08-01 12:01:01 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=11061) INFO 08-01 12:01:01 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep50... +(EngineCore pid=11061) INFO 08-01 12:01:03 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=11061) INFO 08-01 12:01:03 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=11061) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=14078) INFO 08-01 12:15:19 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.27:49229 backend=nccl +(EngineCore pid=14078) INFO 08-01 12:15:19 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=14078) INFO 08-01 12:15:19 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep50... +(EngineCore pid=14078) INFO 08-01 12:15:20 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=14078) INFO 08-01 12:15:20 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=14078) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=2044246) INFO 07-31 01:15:52 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:37821 backend=nccl +(EngineCore pid=2044246) INFO 07-31 01:15:53 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=2044246) INFO 07-31 01:15:53 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep50... +(EngineCore pid=2044246) INFO 07-31 01:15:54 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=2044246) INFO 07-31 01:15:54 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=2044246) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=202106) INFO 08-02 06:54:21 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.27:42471 backend=nccl +(EngineCore pid=202106) INFO 08-02 06:54:21 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=202106) INFO 08-02 06:54:22 [gpu_model_runner.py:4735] Starting to load model outputs/healed/keep50_warmup_fixed_s1224/step0150... +(EngineCore pid=202106) INFO 08-02 06:54:23 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=202106) INFO 08-02 06:54:23 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=202106) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=254118) INFO 08-02 11:59:33 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.27:39475 backend=nccl +(EngineCore pid=254118) INFO 08-02 11:59:33 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=254118) INFO 08-02 11:59:34 [gpu_model_runner.py:4735] Starting to load model outputs/healed/opd_warm_fixed_keep50/step0120... +(EngineCore pid=254118) INFO 08-02 11:59:35 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=254118) INFO 08-02 11:59:35 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=254118) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=148553) INFO 08-02 00:18:05 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.27:35823 backend=nccl +(EngineCore pid=148553) INFO 08-02 00:18:05 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=148553) INFO 08-02 00:18:05 [gpu_model_runner.py:4735] Starting to load model outputs/healed/keep50_offpolicy_warmup_s1224/step0150... +(EngineCore pid=148553) INFO 08-02 00:18:06 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=148553) INFO 08-02 00:18:06 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=148553) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00 outputs/evals/policy_confirm/off_forward_seed1226.json +2026-07-14T15:17:37-07:00 phase1 complete -> outputs/evals/policy_confirm/off_forward_seed1226.json +2026-07-14T15:17:37-07:00 phase2: waiting for tmux policy-pair-1225 to end +2026-07-14T15:17:37-07:00 GPU GPU-8ca70870-ddf2-d274-bcc5-182c2075bced released (33 MiB) +2026-07-14T15:17:37-07:00 GPU GPU-a6acf07f-31f5-618f-a5d0-c0017e7e2e27 released (46 MiB) +2026-07-14T15:17:37-07:00 GPU GPU-864c54df-0130-7780-e271-8a5551d1733f released (15 MiB) +2026-07-14T15:17:37-07:00 phase2: launching on-policy seed 1226 (teacher GPU-8ca70870-ddf2-d274-bcc5-182c2075bced, student GPU-a6acf07f-31f5-618f-a5d0-c0017e7e2e27, vLLM GPU-864c54df-0130-7780-e271-8a5551d1733f) +2026-07-14T16:37:38-07:00 phase2: validated outputs/healed/policy_confirm/on_reverse_seed1226/step0050 +2026-07-14T16:37:38-07:00 GPU GPU-864c54df-0130-7780-e271-8a5551d1733f released (15 MiB) +{ + "correct": 852, + "accuracy": 0.6459438968915845, + "finished": 1259, + "finish_rate": 0.954510993176649, + "mean_completion_tokens": 119.7407126611069 +} +saved item-level results -> outputs/evals/policy_confirm/on_reverse_seed1226.json +{ + "on_policy": "on_reverse_seed1226", + "off_policy": "off_forward_seed1226", + "frame": "chat", + "n": 1319, + "on_accuracy": 0.6459438968915845, + "off_accuracy": 0.6444275966641395, + "off_minus_on": -0.001516300227445034, + "paired_counts": { + "both_correct": 755, + "on_only": 97, + "off_only": 95, + "both_wrong": 372 + }, + "mcnemar_exact_p": 0.9424925696532238 +} +2026-07-14T16:39:29-07:00 seed 1226 pair complete -> outputs/evals/policy_confirm/pair_seed1226.json diff --git a/healed/policy_confirm/combo.log b/healed/policy_confirm/combo.log new file mode 100644 index 0000000000000000000000000000000000000000..344745b8b3759298620da8ebc062eec4ccd66b22 --- /dev/null +++ b/healed/policy_confirm/combo.log @@ -0,0 +1,35 @@ +2026-07-14T14:58:20-07:00 waiting for tmux policy-chain-1226 to end (confirmatory set first) +2026-07-14T16:40:21-07:00 GPU GPU-8ca70870-ddf2-d274-bcc5-182c2075bced released (33 MiB) +2026-07-14T16:40:21-07:00 GPU GPU-a6acf07f-31f5-618f-a5d0-c0017e7e2e27 released (46 MiB) +2026-07-14T16:40:21-07:00 GPU GPU-864c54df-0130-7780-e271-8a5551d1733f released (15 MiB) +2026-07-14T16:40:21-07:00 combo phase1: off-policy 25 steps, seed 1225 on GPU-864c54df-0130-7780-e271-8a5551d1733f +2026-07-14T16:40:21-07:00 combo phase1: off-policy 25 steps, seed 1224 on GPU-8ca70870-ddf2-d274-bcc5-182c2075bced +2026-07-14T16:58:18-07:00 combo phase1 validated: outputs/healed/policy_confirm/combo_off25_seed1225/step0025 +2026-07-14T16:59:26-07:00 combo phase1 validated: outputs/healed/policy_confirm/combo_off25_seed1224/step0025 +2026-07-14T16:59:26-07:00 GPU GPU-8ca70870-ddf2-d274-bcc5-182c2075bced released (33 MiB) +2026-07-14T16:59:26-07:00 combo phase1: off-policy 25 steps, seed 1226 on GPU-8ca70870-ddf2-d274-bcc5-182c2075bced +2026-07-14T17:16:54-07:00 combo phase1 validated: outputs/healed/policy_confirm/combo_off25_seed1226/step0025 +2026-07-14T17:16:54-07:00 GPU GPU-8ca70870-ddf2-d274-bcc5-182c2075bced released (33 MiB) +2026-07-14T17:16:54-07:00 GPU GPU-864c54df-0130-7780-e271-8a5551d1733f released (15 MiB) +2026-07-14T17:16:54-07:00 combo phase2: on-policy steps 26-50, seed 1224 +2026-07-14T17:57:46-07:00 combo phase2 validated: outputs/healed/policy_confirm/combo_on25_seed1224/step0050 +2026-07-14T17:57:46-07:00 GPU GPU-8ca70870-ddf2-d274-bcc5-182c2075bced released (33 MiB) +2026-07-14T17:57:46-07:00 GPU GPU-a6acf07f-31f5-618f-a5d0-c0017e7e2e27 released (46 MiB) +2026-07-14T17:57:46-07:00 GPU GPU-864c54df-0130-7780-e271-8a5551d1733f released (15 MiB) +2026-07-14T17:59:35-07:00 combo seed 1224 complete -> outputs/evals/policy_confirm/combo_off25on25_seed1224.json +2026-07-14T17:59:35-07:00 GPU GPU-864c54df-0130-7780-e271-8a5551d1733f released (15 MiB) +2026-07-14T17:59:35-07:00 combo phase2: on-policy steps 26-50, seed 1225 +2026-07-14T18:40:52-07:00 combo phase2 validated: outputs/healed/policy_confirm/combo_on25_seed1225/step0050 +2026-07-14T18:40:52-07:00 GPU GPU-8ca70870-ddf2-d274-bcc5-182c2075bced released (33 MiB) +2026-07-14T18:40:52-07:00 GPU GPU-a6acf07f-31f5-618f-a5d0-c0017e7e2e27 released (46 MiB) +2026-07-14T18:40:52-07:00 GPU GPU-864c54df-0130-7780-e271-8a5551d1733f released (15 MiB) +2026-07-14T18:42:42-07:00 combo seed 1225 complete -> outputs/evals/policy_confirm/combo_off25on25_seed1225.json +2026-07-14T18:42:42-07:00 GPU GPU-864c54df-0130-7780-e271-8a5551d1733f released (15 MiB) +2026-07-14T18:42:42-07:00 combo phase2: on-policy steps 26-50, seed 1226 +2026-07-14T19:24:14-07:00 combo phase2 validated: outputs/healed/policy_confirm/combo_on25_seed1226/step0050 +2026-07-14T19:24:14-07:00 GPU GPU-8ca70870-ddf2-d274-bcc5-182c2075bced released (33 MiB) +2026-07-14T19:24:14-07:00 GPU GPU-a6acf07f-31f5-618f-a5d0-c0017e7e2e27 released (46 MiB) +2026-07-14T19:24:14-07:00 GPU GPU-864c54df-0130-7780-e271-8a5551d1733f released (15 MiB) +2026-07-14T19:26:03-07:00 combo seed 1226 complete -> outputs/evals/policy_confirm/combo_off25on25_seed1226.json +2026-07-14T19:26:03-07:00 GPU GPU-864c54df-0130-7780-e271-8a5551d1733f released (15 MiB) +2026-07-14T19:26:03-07:00 combo arm complete for all seeds diff --git a/healed/policy_confirm/combo_off25_seed1224.console.log b/healed/policy_confirm/combo_off25_seed1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..e4fbe578a5fa16743df1ab86454d9fa47b06ce1e --- /dev/null +++ b/healed/policy_confirm/combo_off25_seed1224.console.log @@ -0,0 +1,75 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/combo_off25_seed1224/wandb/run-20260714_164028-fnb5m9as +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-combo-off25-seed1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/fnb5m9as + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/policy_confirm/combo_off25_seed1224/step0025 +wandb: updating run metadata +wandb: uploading summary, console lines 31-31 +wandb: +wandb: Run history: +wandb: comp_len ▃▆▆█▄▅█▇▄▄▄▂▇▇▅▂▄▂▄▅▆▃▁▄▄ +wandb: cumulative_loss_tokens ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate █▄▅▄▅▅▁▃▇▆▇▇▅▃▅█▄▇▆▆▄▇█▆▅ +wandb: forward_topk_kl ▆██▆▃▇▃▄▂▃▂▂▂▃▂▂▂▁▂▁▁▁▁▁▁ +wandb: grad_norm ██▇▄▃▄▂▂▂▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▃▄▅▅▆▇█████████████████ +wandb: mem_gb ▁▇▄▆▅▇█▇▆█▅▆▆▆█▃▇▇▄▅▆▅▂▆▇ +wandb: step ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 526.3 +wandb: cumulative_loss_tokens 3000000 +wandb: epoch 0 +wandb: finish_rate 0.838 +wandb: forward_topk_kl 0.07561 +wandb: grad_norm 0.38086 +wandb: lr 3e-05 +wandb: mem_gb 16.05 +wandb: step 25 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run policy-combo-off25-seed1224 at: https://wandb.ai/hbfreed/glean-heal/runs/fnb5m9as +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/combo_off25_seed1224/wandb/run-20260714_164028-fnb5m9as/logs diff --git a/healed/policy_confirm/combo_off25_seed1225.console.log b/healed/policy_confirm/combo_off25_seed1225.console.log new file mode 100644 index 0000000000000000000000000000000000000000..5fa90b2e5d1a67c475689d71a8f202089f674f89 --- /dev/null +++ b/healed/policy_confirm/combo_off25_seed1225.console.log @@ -0,0 +1,77 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run 5fsq79h5 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/combo_off25_seed1225/wandb/run-20260714_164028-5fsq79h5 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-combo-off25-seed1225 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/5fsq79h5 + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/policy_confirm/combo_off25_seed1225/step0025 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml; uploading summary, console lines 31-31 +wandb: uploading data +wandb: +wandb: Run history: +wandb: comp_len █▅▆▆▅▂▆▃▃▃▃▂▄▃▁▆▅▄▅▃▃▆▄▃▆ +wandb: cumulative_loss_tokens ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate ▂▆▃▃▄█▁▇▆▆▇▆▆▇█▃▃▄▄▆▅▂▄▆▃ +wandb: forward_topk_kl ▇▆█▇▄▃▃▂▃▂▂▂▂▁▁▂▂▁▂▁▁▁▁▁▁ +wandb: grad_norm █▇▇▄▃▃▂▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▃▄▅▅▆▇█████████████████ +wandb: mem_gb ▃▅▅▅▅▃▆▅▃▃▁▄▃▅▅▇▆▅▅▅▆▆█▅▅ +wandb: step ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: t_data_s ▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 576.9 +wandb: cumulative_loss_tokens 3000000 +wandb: epoch 0 +wandb: finish_rate 0.764 +wandb: forward_topk_kl 0.07792 +wandb: grad_norm 0.38086 +wandb: lr 3e-05 +wandb: mem_gb 16.04 +wandb: step 25 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run policy-combo-off25-seed1225 at: https://wandb.ai/hbfreed/glean-heal/runs/5fsq79h5 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/combo_off25_seed1225/wandb/run-20260714_164028-5fsq79h5/logs diff --git a/healed/policy_confirm/combo_off25_seed1226.console.log b/healed/policy_confirm/combo_off25_seed1226.console.log new file mode 100644 index 0000000000000000000000000000000000000000..03ab7e1f7f773006eb441893db6ef6554338c472 --- /dev/null +++ b/healed/policy_confirm/combo_off25_seed1226.console.log @@ -0,0 +1,77 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run jh6zgcxt +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/combo_off25_seed1226/wandb/run-20260714_165932-jh6zgcxt +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-combo-off25-seed1226 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/jh6zgcxt + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/policy_confirm/combo_off25_seed1226/step0025 +wandb: updating run metadata +wandb: uploading config.yaml; uploading output.log; uploading wandb-summary.json +wandb: uploading summary +wandb: +wandb: Run history: +wandb: comp_len ▂▃▆▅█▁▆▇▇▄▆▂▂▃█▇▇▆▄▇▇▃▃▃▇ +wandb: cumulative_loss_tokens ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate ▇▆▂▆▂█▄▃▂▇▄█▇▇▁▃▂▁▅▁▂▄▇▇▂ +wandb: forward_topk_kl ▇██▅▆▃▄▃▂▂▂▁▂▂▂▄▂▂▁▁▂▁▁▁▂ +wandb: grad_norm █▇▆▄▃▂▂▂▂▁▁▁▁▁▁▂▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▃▄▅▅▆▇█████████████████ +wandb: mem_gb ▁▅▆▇▆▂▅▅▄▄▅▄▂▅▆▆▅█▅▅▆▆▅▂▆ +wandb: step ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: t_data_s ▁█▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 576.9 +wandb: cumulative_loss_tokens 3000000 +wandb: epoch 0 +wandb: finish_rate 0.745 +wandb: forward_topk_kl 0.08893 +wandb: grad_norm 0.44141 +wandb: lr 3e-05 +wandb: mem_gb 16.06 +wandb: step 25 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run policy-combo-off25-seed1226 at: https://wandb.ai/hbfreed/glean-heal/runs/jh6zgcxt +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/combo_off25_seed1226/wandb/run-20260714_165932-jh6zgcxt/logs diff --git a/healed/policy_confirm/combo_on25_seed1224.console.log b/healed/policy_confirm/combo_on25_seed1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..766de0ca08295192ba74f8a8f4b6e92e6e4c653e --- /dev/null +++ b/healed/policy_confirm/combo_on25_seed1224.console.log @@ -0,0 +1,94 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/combo_on25_seed1224/wandb/run-20260714_171701-62dayi8c +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-combo-on25-seed1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/62dayi8c + Loading checkpoint shards: 0%| | 0/3 [00:00 63998 prompts + Filter: 0%| | 0/63998 [00:00 63998 prompts +WARNING: difficulty filter removed nothing — the slice likely carries difficulty=None (Dolci math sources do), so the teacher-competence guard is NOT in effect + Map: 0%| | 0/63998 [00:00 outputs/healed/policy_confirm/combo_on25_seed1224/step0050 +wandb: updating run metadata +wandb: uploading wandb-summary.json; uploading config.yaml; uploading output.log +wandb: +wandb: Run history: +wandb: comp_len ▄█▅▄▃▆▄▁▄▁▄▆▆▄▆▃▅▅▂▄▆▄▃█▄ +wandb: cumulative_loss_tokens ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate ▃▂▃▅█▆▄▆█▆▅▅▁▅▅▅▃▇█▃▄▆▄▁▃ +wandb: grad_norm █▄▃▂▂▂▃▃▁▂▂▂▂▂▁▄▂▁▃▂▁▂▃▁▂ +wandb: lr ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: mem_gb ▃▇▅▅▄▄▅▅▆▅▅▃▆▄▄█▄▂▄▁▃▂▃▅▄ +wandb: reverse_kl █▇▅▆▃▁▂█▂▅▁▂▂▆▁▁▃▅▁▁▄▂▃▁▂ +wandb: step ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +4 ... +wandb: +wandb: Run summary: +wandb: comp_len 489.8 +wandb: cumulative_loss_tokens 6000000 +wandb: epoch 0 +wandb: finish_rate 0.147 +wandb: grad_norm 0.98828 +wandb: lr 3e-05 +wandb: mem_gb 16.53 +wandb: reverse_kl 0.10983 +wandb: step 50 +wandb: t_data_s 0 +wandb: +5 ... +wandb: +wandb: 🚀 View run policy-combo-on25-seed1224 at: https://wandb.ai/hbfreed/glean-heal/runs/62dayi8c +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/combo_on25_seed1224/wandb/run-20260714_171701-62dayi8c/logs +{ + "correct": 851, + "accuracy": 0.645185746777862, + "finished": 1290, + "finish_rate": 0.978013646702047, + "mean_completion_tokens": 115.40864291129644 +} +saved item-level results -> outputs/evals/policy_confirm/combo_off25on25_seed1224.json diff --git a/healed/policy_confirm/combo_on25_seed1225.console.log b/healed/policy_confirm/combo_on25_seed1225.console.log new file mode 100644 index 0000000000000000000000000000000000000000..97d8ebe7c27560c33e6512da06267e9a2b538a78 --- /dev/null +++ b/healed/policy_confirm/combo_on25_seed1225.console.log @@ -0,0 +1,95 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run uctw2344 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/combo_on25_seed1225/wandb/run-20260714_175941-uctw2344 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-combo-on25-seed1225 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/uctw2344 + Loading checkpoint shards: 0%| | 0/3 [00:00 63998 prompts + Filter: 0%| | 0/63998 [00:00 63998 prompts +WARNING: difficulty filter removed nothing — the slice likely carries difficulty=None (Dolci math sources do), so the teacher-competence guard is NOT in effect + Map: 0%| | 0/63998 [00:00 outputs/healed/policy_confirm/combo_on25_seed1225/step0050 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▅▆▅▆▄▅▅▄▆▅▄█▃▄▄▄▁▄▄▃▅▄▅▃▂ +wandb: cumulative_loss_tokens ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate ▄▂▅▁▇▄▅▆▃▃▃▁▆▅▆▅█▆▆▇▅▅▅▅█ +wandb: grad_norm ▇▄▂▂▃▂▄▂▂▁▃▁▂▂▃▂▂▂█▂▂▂▂▂▃ +wandb: lr ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: mem_gb ▁▅▃▄▆▃▃▅█▃▅▆▆▄▆▅▇▃▄▃▄▃▇▅▃ +wandb: reverse_kl ██▅▄▅▃▃▃▄▃▆▃▁▃▅▃▃▂▂▂▂▁▃▂▁ +wandb: step ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +4 ... +wandb: +wandb: Run summary: +wandb: comp_len 480 +wandb: cumulative_loss_tokens 6000000 +wandb: epoch 0 +wandb: finish_rate 0.224 +wandb: grad_norm 1.01562 +wandb: lr 3e-05 +wandb: mem_gb 16.46 +wandb: reverse_kl 0.10267 +wandb: step 50 +wandb: t_data_s 0 +wandb: +5 ... +wandb: +wandb: 🚀 View run policy-combo-on25-seed1225 at: https://wandb.ai/hbfreed/glean-heal/runs/uctw2344 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/combo_on25_seed1225/wandb/run-20260714_175941-uctw2344/logs +{ + "correct": 873, + "accuracy": 0.6618650492797574, + "finished": 1270, + "finish_rate": 0.9628506444275967, + "mean_completion_tokens": 118.31159969673996 +} +saved item-level results -> outputs/evals/policy_confirm/combo_off25on25_seed1225.json diff --git a/healed/policy_confirm/combo_on25_seed1226.console.log b/healed/policy_confirm/combo_on25_seed1226.console.log new file mode 100644 index 0000000000000000000000000000000000000000..f8c8b416cbb186679c06dc463c359c861662be07 --- /dev/null +++ b/healed/policy_confirm/combo_on25_seed1226.console.log @@ -0,0 +1,95 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run m81lriv3 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/combo_on25_seed1226/wandb/run-20260714_184248-m81lriv3 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-combo-on25-seed1226 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/m81lriv3 + Loading checkpoint shards: 0%| | 0/3 [00:00 63998 prompts + Filter: 0%| | 0/63998 [00:00 63998 prompts +WARNING: difficulty filter removed nothing — the slice likely carries difficulty=None (Dolci math sources do), so the teacher-competence guard is NOT in effect + Map: 0%| | 0/63998 [00:00 outputs/healed/policy_confirm/combo_on25_seed1226/step0050 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▄▇▄▇▇▆▅▅▅▅▆▄█▅▂▇▅▇▇▇▆▁▅▄▆ +wandb: cumulative_loss_tokens ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate ▅▃▄▁▃▃▅▄▅▃▃▅▂▃▆▂▄▂▃▂▄█▄▄▂ +wandb: grad_norm █▄▁▂▃▂▂▂▁▃▃▂▂▂▂▂▂▂▁▃▁▂▁▂▁ +wandb: lr ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: mem_gb ▃▂▂▇▁▃▄▂▅▄▁▆▅▁▁▂▄▆▅▁█▃▁▅▃ +wandb: reverse_kl █▇▄▅▃▂▃▃▃▅▅▅▂▃▁▁▂▂▂▁▁▁▁▁▂ +wandb: step ▁▁▂▂▂▂▃▃▃▄▄▄▅▅▅▅▆▆▆▇▇▇▇██ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +4 ... +wandb: +wandb: Run summary: +wandb: comp_len 489.8 +wandb: cumulative_loss_tokens 6000000 +wandb: epoch 0 +wandb: finish_rate 0.155 +wandb: grad_norm 0.85156 +wandb: lr 3e-05 +wandb: mem_gb 16.48 +wandb: reverse_kl 0.10447 +wandb: step 50 +wandb: t_data_s 0 +wandb: +5 ... +wandb: +wandb: 🚀 View run policy-combo-on25-seed1226 at: https://wandb.ai/hbfreed/glean-heal/runs/m81lriv3 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/combo_on25_seed1226/wandb/run-20260714_184248-m81lriv3/logs +{ + "correct": 851, + "accuracy": 0.645185746777862, + "finished": 1279, + "finish_rate": 0.9696739954510993, + "mean_completion_tokens": 116.73161485974222 +} +saved item-level results -> outputs/evals/policy_confirm/combo_off25on25_seed1226.json diff --git a/healed/policy_confirm/off_forward_lr1e-4_seed1224.console.log b/healed/policy_confirm/off_forward_lr1e-4_seed1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..279652228c5d19d3bbaf7e6204d0ead6b5570080 --- /dev/null +++ b/healed/policy_confirm/off_forward_lr1e-4_seed1224.console.log @@ -0,0 +1,104 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run 1mesnn32 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/off_forward_lr1e-4_seed1224/wandb/run-20260714_110703-1mesnn32 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-off-forward-lr1e-4-seed1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/1mesnn32 + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/policy_confirm/off_forward_lr1e-4_seed1224/step0050 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▃▆▅▇▄▇▆▄▄▄▆▆▅▂▄▄▄▆▃▁▃▃█▃▅▅▇▄▅▆▅▅▃▂▃▅▄▆▁▆ +wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇███ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate █▄▅▄▅▁▃▇▆▇▅▃▅█▄▆▆▄▇█▅▄▆▂▆▄▇▅▃▆▃▇▆▇▅▅▆▃▇▃ +wandb: forward_topk_kl ██▆▃▂▄▅▃▃▃▃▆▄▂▃▃▂▃▂▂▃▃▂▄▃▂▂▂▂▃▁▁▁▂▂▁▁▄▁▃ +wandb: grad_norm █▅▄▂▁▂▃▂▂▁▁▂▁▁▁▁▁▁▁▁▁▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▂▁▁ +wandb: lr ▁▃▄▅▆███████████████████████████████████ +wandb: mem_gb ▁▆▅▅▆▆▆▇▆▆▇▃▆▆▄▅▄▂▆▆▆█▃▄▆▆▆▇▇▆▇▆▅▃█▆▆▆▅▇ +wandb: step ▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇██ +wandb: t_data_s ▁█▁▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 563.4 +wandb: cumulative_loss_tokens 6000000 +wandb: epoch 0 +wandb: finish_rate 0.77 +wandb: forward_topk_kl 0.12091 +wandb: grad_norm 0.42578 +wandb: lr 0.0001 +wandb: mem_gb 16.09 +wandb: step 50 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run policy-off-forward-lr1e-4-seed1224 at: https://wandb.ai/hbfreed/glean-heal/runs/1mesnn32 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/off_forward_lr1e-4_seed1224/wandb/run-20260714_110703-1mesnn32/logs diff --git a/healed/policy_confirm/off_forward_lr1e-5_seed1224.console.log b/healed/policy_confirm/off_forward_lr1e-5_seed1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..abbf9f0be7dc34a2d6acd87edb8e869525530262 --- /dev/null +++ b/healed/policy_confirm/off_forward_lr1e-5_seed1224.console.log @@ -0,0 +1,104 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run 9q7n1667 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/off_forward_lr1e-5_seed1224/wandb/run-20260714_095410-9q7n1667 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-off-forward-lr1e-5-seed1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/9q7n1667 + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/policy_confirm/off_forward_lr1e-5_seed1224/step0050 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▃▅▅▇▄▇▆▄▄▄▆▆▅▂▄▄▆▃▁▄▃▃█▃▄▃▅▇▃▅▂▅▅▃▂▇▅▄▅▆ +wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇███ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate █▃▅▄▅▃▇▆▇▇▂▄█▃▇▅▄▆█▆▄▆▁▆▅▅▂▅▃▃▄▆▇▆▅▄▆▂▇▃ +wandb: forward_topk_kl ▅▇█▇▅▅▇▄▄▃▃▃▃▂▂▂▂▂▂▁▁▃▂▂▂▁▂▁▂▂▁▁▁▁▁▁▁▂▁▂ +wandb: grad_norm ▇██▇▆▄▄▃▃▂▂▂▁▁▁▁▁▁▁▁▁▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▃▄▅▆▇█████████████████████████████████ +wandb: mem_gb ▁▆▄▅▅▇▆▆▇▄▆▆▇▃▆▅▅▄▂▆▆▆█▃▄▆▆▆▇▇▆▇▆▃█▆▆▆▅▇ +wandb: step ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇███ +wandb: t_data_s ▁█▁▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 563.4 +wandb: cumulative_loss_tokens 6000000 +wandb: epoch 0 +wandb: finish_rate 0.77 +wandb: forward_topk_kl 0.09395 +wandb: grad_norm 0.48242 +wandb: lr 1e-05 +wandb: mem_gb 16.09 +wandb: step 50 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run policy-off-forward-lr1e-5-seed1224 at: https://wandb.ai/hbfreed/glean-heal/runs/9q7n1667 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/off_forward_lr1e-5_seed1224/wandb/run-20260714_095410-9q7n1667/logs diff --git a/healed/policy_confirm/off_forward_lr6e-5_seed1224.console.log b/healed/policy_confirm/off_forward_lr6e-5_seed1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..404388532f5f7b8e5c29df847fb92e91e8e147d5 --- /dev/null +++ b/healed/policy_confirm/off_forward_lr6e-5_seed1224.console.log @@ -0,0 +1,103 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/off_forward_lr6e-5_seed1224/wandb/run-20260714_103036-brk0720b +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-off-forward-lr6e-5-seed1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/brk0720b + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/policy_confirm/off_forward_lr6e-5_seed1224/step0050 +wandb: updating run metadata +wandb: uploading summary, console lines 59-59 +wandb: +wandb: Run history: +wandb: comp_len ▃▅▇▄▅▆▄▄▄▂▅▂▄▂▄▆▃▁▄▄▃█▃▄▅▅▇▄▅▆▅▅▃▂▃▅▄▆▁▆ +wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇███ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate █▄▅▄▅▁▃▇▆▇▅▃▅█▄▆▆▄▇█▅▄▆▂▆▄▇▅▃▆▃▇▄▆▇▅▂▆▇▃ +wandb: forward_topk_kl ▇█▇▅▃▄▄▃▃▂▄▃▂▃▂▂▂▂▂▂▂▂▃▂▃▂▁▂▂▂▁▁▁▁▁▁▁▂▁▂ +wandb: grad_norm █▇▅▃▂▂▂▂▂▁▂▁▁▁▁▁▁▁▁▁▁▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▃▄▅▆▇█████████████████████████████████ +wandb: mem_gb ▁▆▄▅▅▇▆▆▇▄▆▆▇▃▆▄▅▅▄▂▆▆▆█▃▆▆▆▆▇▆▆▇▆▅█▆▆▆▇ +wandb: step ▁▁▁▁▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇███ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 563.4 +wandb: cumulative_loss_tokens 6000000 +wandb: epoch 0 +wandb: finish_rate 0.77 +wandb: forward_topk_kl 0.0815 +wandb: grad_norm 0.38477 +wandb: lr 6e-05 +wandb: mem_gb 16.09 +wandb: step 50 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run policy-off-forward-lr6e-5-seed1224 at: https://wandb.ai/hbfreed/glean-heal/runs/brk0720b +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/off_forward_lr6e-5_seed1224/wandb/run-20260714_103036-brk0720b/logs diff --git a/healed/policy_confirm/off_forward_seed1224.console.log b/healed/policy_confirm/off_forward_seed1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..0454d795647ee4a33b93be52e7465549d3292b06 --- /dev/null +++ b/healed/policy_confirm/off_forward_seed1224.console.log @@ -0,0 +1,103 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/off_forward_seed1224/wandb/run-20260714_091542-mjjns12j +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-off-forward-seed1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/mjjns12j + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/policy_confirm/off_forward_seed1224/step0050 +wandb: updating run metadata +wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml +wandb: +wandb: Run history: +wandb: comp_len ▃▆▅▇▄▆▄▄▄▂▅▂▄▂▄▆▃▁▄▄▃█▃▄▅▅▇▄▅▆▅▅▃▂▃▅▄▆▁▆ +wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇███ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate █▄▅▄▅▁▃▇▆▇▅▃▅█▄▆▆▄▇█▅▄▆▂▆▄▇▅▃▆▃▇▆▇▅▅▆▃▇▃ +wandb: forward_topk_kl ▆██▆▄▄▄▃▃▂▂▃▂▂▂▂▂▂▂▂▂▂▁▂▂▁▁▁▁▂▂▁▁▁▁▁▁▂▁▂ +wandb: grad_norm ██▇▄▃▂▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▂▁▁ +wandb: lr ▁▂▃▄▅▆▇█████████████████████████████████ +wandb: mem_gb ▁▆▄▅▅▇▆▆▇▄▆▆▇▃▇▄▅▅▄▂▆▆█▃▄▆▆▆▇▇▆▇▆▅▃▆▆▆▆▇ +wandb: step ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇███ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 563.4 +wandb: cumulative_loss_tokens 6000000 +wandb: epoch 0 +wandb: finish_rate 0.77 +wandb: forward_topk_kl 0.0735 +wandb: grad_norm 0.41406 +wandb: lr 3e-05 +wandb: mem_gb 16.09 +wandb: step 50 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run policy-off-forward-seed1224 at: https://wandb.ai/hbfreed/glean-heal/runs/mjjns12j +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/off_forward_seed1224/wandb/run-20260714_091542-mjjns12j/logs diff --git a/healed/policy_confirm/off_forward_seed1225.console.log b/healed/policy_confirm/off_forward_seed1225.console.log new file mode 100644 index 0000000000000000000000000000000000000000..5d980c8c1e90324ca3991ca1dc6f74c6b7e3581b --- /dev/null +++ b/healed/policy_confirm/off_forward_seed1225.console.log @@ -0,0 +1,104 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run g5s4cpba +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/off_forward_seed1225/wandb/run-20260714_143756-g5s4cpba +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-off-forward-seed1225 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/g5s4cpba + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/policy_confirm/off_forward_seed1225/step0050 +wandb: updating run metadata +wandb: uploading summary, console lines 59-59 +wandb: +wandb: Run history: +wandb: comp_len █▅▆▆▅▆▄▄▃▃▄▄▂▆▆▆▃▃▆▅▆▄▅▄▅▅▆▄▄▆▆▅▃▅▃▃▅▁▃▅ +wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate ▂▅▃▂▄▆▅▆▆▆▇█▃▃▄▆▅▁▄▆▅▅▄▅▁▃█▆▃▄▆▅▇▁▆▄█▆▅▄ +wandb: forward_topk_kl ▇▆█▄▃▃▃▂▃▂▂▃▂▂▂▂▂▂▂▂▂▁▁▂▁▁▁▂▁▂▁▁▁▁▁▁▁▁▁▁ +wandb: grad_norm █▇▇▄▃▂▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▃▄▅▆▇█████████████████████████████████ +wandb: mem_gb ▄▆▆▅▆▆▆▄▄▂▄▆▆▇▆▅▆▆▆█▄▅▆▆▇▄▄▆▇▁▆▆▅▃▅▂▁▆▄▄ +wandb: step ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇███ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 553 +wandb: cumulative_loss_tokens 6000000 +wandb: epoch 0 +wandb: finish_rate 0.811 +wandb: forward_topk_kl 0.05353 +wandb: grad_norm 0.33203 +wandb: lr 3e-05 +wandb: mem_gb 15.95 +wandb: step 50 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run policy-off-forward-seed1225 at: https://wandb.ai/hbfreed/glean-heal/runs/g5s4cpba +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/off_forward_seed1225/wandb/run-20260714_143756-g5s4cpba/logs diff --git a/healed/policy_confirm/off_forward_seed1226.console.log b/healed/policy_confirm/off_forward_seed1226.console.log new file mode 100644 index 0000000000000000000000000000000000000000..710bd9315a9b385d6e9c93056e9b3822dcde2ed5 --- /dev/null +++ b/healed/policy_confirm/off_forward_seed1226.console.log @@ -0,0 +1,104 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run nkvwlrbd +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/off_forward_seed1226/wandb/run-20260714_143758-nkvwlrbd +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-off-forward-seed1226 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/nkvwlrbd + Loading checkpoint shards: 0%| | 0/2 [00:00 outputs/healed/policy_confirm/off_forward_seed1226/step0050 +wandb: updating run metadata +wandb: uploading config.yaml; uploading output.log; uploading wandb-summary.json +wandb: +wandb: Run history: +wandb: comp_len ▂▃▆█▁▇▇▄▆▂▃█▇▇▆▇▇▃▃▃▅▆▄▆▅▅▆▄▆▂▅▅▄▆▆▅▅▅▅▆ +wandb: cumulative_loss_tokens ▁▁▁▂▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇██ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate █▇▂▆▂▄▃▂█▅▇▇▁▃▁▁▂▅▇█▄▃▆▄▅▆▆▃▆▂▅▄▆▄▅▃▄▅▄▃ +wandb: forward_topk_kl ▇██▆▆▄▃▃▂▂▂▂▃▅▂▂▂▃▂▂▂▂▂▂▂▁▁▁▂▂▁▁▂▁▁▁▁▁▁▁ +wandb: grad_norm █▇▆▄▃▂▂▂▁▁▁▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: lr ▁▂▃▄▅▆▇█████████████████████████████████ +wandb: mem_gb ▁▅▆▇▆▅▅▄▅▄▆▆▅█▅▆▆▅▂▆▆▅▆▃▆▅▆▅▄▃▅▅▆▅▃▅▆▆▆▄ +wandb: step ▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇██ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁█▁ +wandb: +3 ... +wandb: +wandb: Run summary: +wandb: comp_len 563.4 +wandb: cumulative_loss_tokens 6000000 +wandb: epoch 0 +wandb: finish_rate 0.784 +wandb: forward_topk_kl 0.05677 +wandb: grad_norm 0.35547 +wandb: lr 3e-05 +wandb: mem_gb 15.97 +wandb: step 50 +wandb: t_data_s 0 +wandb: +4 ... +wandb: +wandb: 🚀 View run policy-off-forward-seed1226 at: https://wandb.ai/hbfreed/glean-heal/runs/nkvwlrbd +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/off_forward_seed1226/wandb/run-20260714_143758-nkvwlrbd/logs diff --git a/healed/policy_confirm/off_lr_search_seed1224.log b/healed/policy_confirm/off_lr_search_seed1224.log new file mode 100644 index 0000000000000000000000000000000000000000..b8ba761e7c107aa340788b1f017a81a465414361 --- /dev/null +++ b/healed/policy_confirm/off_lr_search_seed1224.log @@ -0,0 +1,38 @@ +2026-07-14T08:13:31-07:00 waiting for primary pair session policy-pair-1224 +2026-07-14T09:54:03-07:00 GPU 2 released (15 MiB) +2026-07-14T09:54:03-07:00 starting off-policy seed 1224, lr=1e-5 +2026-07-14T10:28:43-07:00 GPU 2 released (15 MiB) +{ + "correct": 833, + "accuracy": 0.6315390447308568, + "finished": 1300, + "finish_rate": 0.9855951478392722, + "mean_completion_tokens": 114.62016679302502 +} +saved item-level results -> outputs/evals/policy_confirm/off_forward_lr1e-5_seed1224.json +2026-07-14T10:30:30-07:00 GPU 2 released (15 MiB) +2026-07-14T10:30:30-07:00 starting off-policy seed 1224, lr=6e-5 +2026-07-14T11:05:05-07:00 GPU 2 released (15 MiB) +{ + "correct": 812, + "accuracy": 0.6156178923426838, + "finished": 1261, + "finish_rate": 0.956027293404094, + "mean_completion_tokens": 122.13646702047005 +} +saved item-level results -> outputs/evals/policy_confirm/off_forward_lr6e-5_seed1224.json +2026-07-14T11:06:57-07:00 GPU 2 released (15 MiB) +2026-07-14T11:06:57-07:00 starting off-policy seed 1224, lr=1e-4 +2026-07-14T11:41:29-07:00 GPU 2 released (15 MiB) +{ + "correct": 713, + "accuracy": 0.5405610310841547, + "finished": 1199, + "finish_rate": 0.909021986353298, + "mean_completion_tokens": 132.58908263836238 +} +saved item-level results -> outputs/evals/policy_confirm/off_forward_lr1e-4_seed1224.json +2026-07-14T11:43:25-07:00 GPU 2 released (15 MiB) +jq: error: syntax error, unexpected ',', expecting ':' (Unix shell quoting issues?) at , line 9: + label, +jq: 1 compile error diff --git a/healed/policy_confirm/on_reverse_seed1224.console.log b/healed/policy_confirm/on_reverse_seed1224.console.log new file mode 100644 index 0000000000000000000000000000000000000000..26da3448711848a9b3070686db92bea4d378e762 --- /dev/null +++ b/healed/policy_confirm/on_reverse_seed1224.console.log @@ -0,0 +1,113 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run vy0sblk9 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/on_reverse_seed1224/wandb/run-20260714_075432-vy0sblk9 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-on-reverse-seed1224 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/vy0sblk9 + Loading checkpoint shards: 0%| | 0/3 [00:00 63998 prompts + Filter: 0%| | 0/63998 [00:00 63998 prompts +WARNING: difficulty filter removed nothing — the slice likely carries difficulty=None (Dolci math sources do), so the teacher-competence guard is NOT in effect + Map: 0%| | 0/63998 [00:00 outputs/healed/policy_confirm/on_reverse_seed1224/step0050 +wandb: updating run metadata +wandb: uploading config.yaml; uploading output.log; uploading wandb-summary.json +wandb: +wandb: Run history: +wandb: comp_len ▄▅▅▇█▅▄▅▃▂▅▅▇▄▄▄▂▂▆▇▇▇█▆▃▆▄▄▄▆▇▇▆▄▅▅▇▁▇▇ +wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇███ +wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: finish_rate ▄▃▅▂▁▅▄▄▄▄▅▄▄▄▂▃█▅▄▃▃▂▁▃▇▄▅▆▄▃▂▃▃▃▅▄▄▅▂▄ +wandb: grad_norm █▇▇▅▄▃▂▂▂▂▁▂▁▂▁▂▂▁▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▂▁▁▁▁▁ +wandb: lr ▁▂▃▄▅▆▇█████████████████████████████████ +wandb: mem_gb ▁▄▅▄▁▄▂▆▂▅▄▄▃▄▃▅▃▃▄▆▇▅▆▄▃▄▅▄▅▃▄▄█▄▂▁█▄▂▆ +wandb: reverse_kl █▇█▇▅▄▃▃▃▃▃▂▂▂▂▂▁▂▁▁▁▁▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: step ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇███ +wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ +wandb: +4 ... +wandb: +wandb: Run summary: +wandb: comp_len 500 +wandb: cumulative_loss_tokens 6000000 +wandb: epoch 0 +wandb: finish_rate 0.15 +wandb: grad_norm 0.99609 +wandb: lr 3e-05 +wandb: mem_gb 16.59 +wandb: reverse_kl 0.10605 +wandb: step 50 +wandb: t_data_s 0 +wandb: +5 ... +wandb: +wandb: 🚀 View run policy-on-reverse-seed1224 at: https://wandb.ai/hbfreed/glean-heal/runs/vy0sblk9 +wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal +wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s) +wandb: Find logs at: outputs/healed/policy_confirm/on_reverse_seed1224/wandb/run-20260714_075432-vy0sblk9/logs diff --git a/healed/policy_confirm/on_reverse_seed1225.console.log b/healed/policy_confirm/on_reverse_seed1225.console.log new file mode 100644 index 0000000000000000000000000000000000000000..6af23234368a552965bdc1640d463b0fbf75a2f7 --- /dev/null +++ b/healed/policy_confirm/on_reverse_seed1225.console.log @@ -0,0 +1,113 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run dwzirwyj +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/on_reverse_seed1225/wandb/run-20260714_131739-dwzirwyj +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-on-reverse-seed1225 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/dwzirwyj + Loading checkpoint shards: 0%| | 0/3 [00:00 63998 prompts + Filter: 0%| | 0/63998 [00:00 63998 prompts +WARNING: difficulty filter removed nothing — the slice likely carries difficulty=None (Dolci math sources do), so the teacher-competence guard is NOT in effect + Map: 0%| | 0/63998 [00:00 outputs/healed/policy_confirm/on_reverse_seed1225/step0050 diff --git a/healed/policy_confirm/on_reverse_seed1226.console.log b/healed/policy_confirm/on_reverse_seed1226.console.log new file mode 100644 index 0000000000000000000000000000000000000000..ae585bf4bec537057b6e48e60508da8c4b8e4a0b --- /dev/null +++ b/healed/policy_confirm/on_reverse_seed1226.console.log @@ -0,0 +1,113 @@ +/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available. + warnings.warn('Grouped GEMM not available.') +wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc. +wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin +wandb: setting up run q74ep8k9 +wandb: Tracking run with wandb version 0.28.0 +wandb: Run data is saved locally in outputs/healed/policy_confirm/on_reverse_seed1226/wandb/run-20260714_151743-q74ep8k9 +wandb: Run `wandb offline` to turn off syncing. +wandb: Syncing run policy-on-reverse-seed1226 +wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal +wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/q74ep8k9 + Loading checkpoint shards: 0%| | 0/3 [00:00 63998 prompts + Filter: 0%| | 0/63998 [00:00 63998 prompts +WARNING: difficulty filter removed nothing — the slice likely carries difficulty=None (Dolci math sources do), so the teacher-competence guard is NOT in effect + Map: 0%| | 0/63998 [00:00 outputs/healed/policy_confirm/on_reverse_seed1226/step0050 diff --git a/healed/policy_confirm/pair_seed1224.log b/healed/policy_confirm/pair_seed1224.log new file mode 100644 index 0000000000000000000000000000000000000000..3b4507fa65741c18de9960037f62cffffcee39c1 --- /dev/null +++ b/healed/policy_confirm/pair_seed1224.log @@ -0,0 +1,39 @@ +2026-07-14T08:07:04-07:00 waiting for tmux policy-on-rkl-1224 +2026-07-14T09:15:35-07:00 validated outputs/healed/policy_confirm/on_reverse_seed1224/step0050 +2026-07-14T09:15:35-07:00 GPU 2 released (15 MiB) +2026-07-14T09:15:35-07:00 starting off-policy seed 1224 +2026-07-14T09:50:10-07:00 validated outputs/healed/policy_confirm/off_forward_seed1224/step0050 +2026-07-14T09:50:10-07:00 GPU 2 released (15 MiB) +{ + "correct": 859, + "accuracy": 0.6512509476876421, + "finished": 1267, + "finish_rate": 0.9605761940864291, + "mean_completion_tokens": 123.09476876421532 +} +saved item-level results -> outputs/evals/policy_confirm/on_reverse_seed1224.json +{ + "correct": 836, + "accuracy": 0.6338134950720242, + "finished": 1291, + "finish_rate": 0.9787717968157695, + "mean_completion_tokens": 116.0803639120546 +} +saved item-level results -> outputs/evals/policy_confirm/off_forward_seed1224.json +{ + "on_policy": "on_reverse_seed1224", + "off_policy": "off_forward_seed1224", + "frame": "chat", + "n": 1319, + "on_accuracy": 0.6512509476876421, + "off_accuracy": 0.6338134950720242, + "off_minus_on": -0.017437452615617893, + "paired_counts": { + "both_correct": 745, + "on_only": 114, + "off_only": 91, + "both_wrong": 369 + }, + "mcnemar_exact_p": 0.12419617466066926 +} +2026-07-14T09:53:51-07:00 seed 1224 pair complete -> outputs/evals/policy_confirm/pair_seed1224.json diff --git a/healed/policy_confirm/pair_seed1225.log b/healed/policy_confirm/pair_seed1225.log new file mode 100644 index 0000000000000000000000000000000000000000..423bc55375ee63522d9b4e8ee059f25eb18cffc6 --- /dev/null +++ b/healed/policy_confirm/pair_seed1225.log @@ -0,0 +1,39 @@ +2026-07-14T13:18:18-07:00 waiting for tmux policy-on-rkl-1225 +2026-07-14T14:37:49-07:00 validated outputs/healed/policy_confirm/on_reverse_seed1225/step0050 +2026-07-14T14:37:49-07:00 GPU 2 released (15 MiB) +2026-07-14T14:37:49-07:00 starting off-policy seed 1225 +2026-07-14T15:12:52-07:00 validated outputs/healed/policy_confirm/off_forward_seed1225/step0050 +2026-07-14T15:12:52-07:00 GPU 2 released (15 MiB) +{ + "correct": 848, + "accuracy": 0.6429112964366944, + "finished": 1268, + "finish_rate": 0.9613343442001516, + "mean_completion_tokens": 122.05231235784686 +} +saved item-level results -> outputs/evals/policy_confirm/on_reverse_seed1225.json +{ + "correct": 830, + "accuracy": 0.6292645943896892, + "finished": 1289, + "finish_rate": 0.9772554965883244, + "mean_completion_tokens": 116.21000758150113 +} +saved item-level results -> outputs/evals/policy_confirm/off_forward_seed1225.json +{ + "on_policy": "on_reverse_seed1225", + "off_policy": "off_forward_seed1225", + "frame": "chat", + "n": 1319, + "on_accuracy": 0.6429112964366944, + "off_accuracy": 0.6292645943896892, + "off_minus_on": -0.013646702047005308, + "paired_counts": { + "both_correct": 739, + "on_only": 109, + "off_only": 91, + "both_wrong": 380 + }, + "mcnemar_exact_p": 0.22924659725971577 +} +2026-07-14T15:16:32-07:00 seed 1225 pair complete -> outputs/evals/policy_confirm/pair_seed1225.json diff --git a/healed/refkl_keep50_s1223/args.json b/healed/refkl_keep50_s1223/args.json new file mode 100644 index 0000000000000000000000000000000000000000..5c87e72674aa73d094bd73839fc4f0fe0a062a78 --- /dev/null +++ b/healed/refkl_keep50_s1223/args.json @@ -0,0 +1,66 @@ +{ + "student": "outputs/pruned/glean-0125inst-math-keep50", + "teacher": "allenai/OLMoE-1B-7B-0125-Instruct", + "training_mode": "on-policy", + "kl_direction": "reverse", + "dataset": "allenai/Dolci-Instruct-RL", + "dataset_sources": null, + "max_difficulty": null, + "trajectories": "outputs/teacher_trajectories/dolci_math_curated.jsonl", + "trajectory_dataset": "allenai/Dolci-Instruct-RL", + "off_policy_frames": "chat", + "off_policy_max_seq_len": 2048, + "topk_targets": null, + "max_loss_tokens": null, + "loss_tokens_per_step": null, + "teacher_device": "cuda:0", + "student_device": "cuda:1", + "lr": 3e-05, + "optimizer": "adamw8bit", + "weight_decay": 0.1, + "epochs": 2, + "prompts_per_step": 256, + "group_size": 4, + "rollout_batch": 64, + "micro_batch": 2, + "max_new_tokens": 2048, + "max_prompt_len": 1024, + "warmup_steps": 10, + "max_grad_norm": 1.0, + "eval_every": 10, + "gsm8k_every": 20, + "gsm8k_n": 256, + "gsm8k_batch": 16, + "gsm8k_max_new_tokens": 1024, + "gsm8k_frames": "chat", + "save_every": 1000, + "out_dir": "outputs/healed/refkl_keep50_s1223", + "sweep": 60, + "wandb": true, + "wandb_project": "glean-heal", + "wandb_run_name": "refkl-b005-keep50-s1223", + "wandb_run_id": null, + "wandb_resume": null, + "wandb_mode": "offline", + "no_wandb_sync": false, + "debug": false, + "resume_from": null, + "start_step": 0, + "no_grad_checkpointing": false, + "seed": 1223, + "no_teacher_overlap": false, + "sync_checkpoints": false, + "rollout_engine": "vllm", + "vllm_gpu": "2", + "vllm_port": 8377, + "vllm_refresh_every": 1, + "vllm_serve_bin": "vllm-plugin/.venv/bin/python", + "vllm_gpu_mem_util": 0.85, + "fast_teacher": true, + "reference_kl_beta": 0.05, + "drop_truncated_rollouts": false, + "vllm_max_model_len": null, + "vllm_refresh_mode": "reload", + "vllm_live_dir": null, + "resolved_kl_direction": "reverse" +} \ No newline at end of file diff --git a/healed/refkl_keep50_s1223/train_log.jsonl b/healed/refkl_keep50_s1223/train_log.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..891848fd96a726bb3821ce36feef05bb44cf8ccd --- /dev/null +++ b/healed/refkl_keep50_s1223/train_log.jsonl @@ -0,0 +1,64 @@ +{"step": 1, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6467847454712417, "tokens": 176766, "cumulative_loss_tokens": 176766, "grad_norm": 4.0625, "lr": 6e-06, "finish_rate": 0.984, "comp_len": 690.5, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 53.0, "t_step_s": 132.6, "t_refresh_s": 0.3, "mem_gb": 10.91} +{"step": 1, "gsm8k_n": 256, "gsm8k_quick_chat": 0.58984375, "t_eval_s": 23.9} +{"step": 2, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5850056688621337, "tokens": 179606, "cumulative_loss_tokens": 356372, "grad_norm": 3.96875, "lr": 9e-06, "finish_rate": 0.977, "comp_len": 701.6, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 53.7, "t_step_s": 129.4, "t_refresh_s": 0.3, "mem_gb": 10.78} +{"step": 3, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6041434134971135, "tokens": 176927, "cumulative_loss_tokens": 533299, "grad_norm": 3.484375, "lr": 1.2e-05, "finish_rate": 0.973, "comp_len": 691.1, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 53.2, "t_step_s": 127.6, "t_refresh_s": 0.3, "mem_gb": 10.97} +{"step": 4, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.532148457294122, "tokens": 157081, "cumulative_loss_tokens": 690380, "grad_norm": 3.328125, "lr": 1.5e-05, "finish_rate": 0.996, "comp_len": 613.6, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 47.5, "t_step_s": 116.8, "t_refresh_s": 0.3, "mem_gb": 10.71} +{"step": 5, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5585909076069838, "tokens": 166920, "cumulative_loss_tokens": 857300, "grad_norm": 3.46875, "lr": 1.8e-05, "finish_rate": 0.992, "comp_len": 652.0, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 49.9, "t_step_s": 123.3, "t_refresh_s": 0.3, "mem_gb": 10.73} +{"step": 6, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.537399649530364, "tokens": 165140, "cumulative_loss_tokens": 1022440, "grad_norm": 2.671875, "lr": 2.1e-05, "finish_rate": 0.98, "comp_len": 645.1, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 52.0, "t_step_s": 125.4, "t_refresh_s": 0.3, "mem_gb": 10.99} +{"step": 7, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4606728149481525, "tokens": 182015, "cumulative_loss_tokens": 1204455, "grad_norm": 1.9140625, "lr": 2.4e-05, "finish_rate": 0.988, "comp_len": 711.0, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 54.9, "t_step_s": 131.0, "t_refresh_s": 0.3, "mem_gb": 10.88} +{"step": 8, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4897024684369616, "tokens": 172373, "cumulative_loss_tokens": 1376828, "grad_norm": 1.4765625, "lr": 2.7000000000000002e-05, "finish_rate": 0.984, "comp_len": 673.3, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 50.3, "t_step_s": 122.5, "t_refresh_s": 0.3, "mem_gb": 10.73} +{"step": 9, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6276105611013946, "tokens": 180743, "cumulative_loss_tokens": 1557571, "grad_norm": 1.5703125, "lr": 3e-05, "finish_rate": 0.984, "comp_len": 706.0, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 54.0, "t_step_s": 127.8, "t_refresh_s": 0.3, "mem_gb": 10.93} +{"step": 10, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5146695579860489, "tokens": 188634, "cumulative_loss_tokens": 1746205, "grad_norm": 0.90234375, "lr": 3e-05, "finish_rate": 0.961, "comp_len": 736.9, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 55.7, "t_step_s": 133.0, "t_refresh_s": 0.3, "mem_gb": 11.0} +{"step": 11, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.511969419231122, "tokens": 182558, "cumulative_loss_tokens": 1928763, "grad_norm": 1.921875, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 713.1, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 54.1, "t_step_s": 130.0, "t_refresh_s": 0.3, "mem_gb": 10.99} +{"step": 12, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5104824433820287, "tokens": 212205, "cumulative_loss_tokens": 2140968, "grad_norm": 2.953125, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 828.9, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 62.3, "t_step_s": 143.7, "t_refresh_s": 0.3, "mem_gb": 10.87} +{"step": 13, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6078012882753199, "tokens": 196831, "cumulative_loss_tokens": 2337799, "grad_norm": 2.859375, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 768.9, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 57.0, "t_step_s": 134.5, "t_refresh_s": 0.3, "mem_gb": 10.91} +{"step": 14, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5939751351371753, "tokens": 192023, "cumulative_loss_tokens": 2529822, "grad_norm": 2.890625, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 750.1, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 55.2, "t_step_s": 132.7, "t_refresh_s": 0.3, "mem_gb": 10.84} +{"step": 15, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5914774450738387, "tokens": 205644, "cumulative_loss_tokens": 2735466, "grad_norm": 1.703125, "lr": 3e-05, "finish_rate": 0.945, "comp_len": 803.3, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 59.6, "t_step_s": 138.7, "t_refresh_s": 0.3, "mem_gb": 10.89} +{"step": 16, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5640858046361867, "tokens": 220403, "cumulative_loss_tokens": 2955869, "grad_norm": 1.5, "lr": 3e-05, "finish_rate": 0.938, "comp_len": 860.9, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 64.2, "t_step_s": 147.1, "t_refresh_s": 0.3, "mem_gb": 10.97} +{"step": 17, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6731741216709219, "tokens": 188795, "cumulative_loss_tokens": 3144664, "grad_norm": 1.5234375, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 737.5, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 55.4, "t_step_s": 131.7, "t_refresh_s": 0.3, "mem_gb": 10.95} +{"step": 18, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5915637352744478, "tokens": 197014, "cumulative_loss_tokens": 3341678, "grad_norm": 1.390625, "lr": 3e-05, "finish_rate": 0.953, "comp_len": 769.6, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 60.2, "t_step_s": 140.1, "t_refresh_s": 0.3, "mem_gb": 10.99} +{"step": 19, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6030616320995973, "tokens": 200077, "cumulative_loss_tokens": 3541755, "grad_norm": 1.359375, "lr": 3e-05, "finish_rate": 0.973, "comp_len": 781.6, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 58.1, "t_step_s": 136.4, "t_refresh_s": 0.3, "mem_gb": 10.8} +{"step": 20, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5272375456274512, "tokens": 214401, "cumulative_loss_tokens": 3756156, "grad_norm": 1.140625, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 837.5, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 66.9, "t_step_s": 149.2, "t_refresh_s": 0.3, "mem_gb": 10.75} +{"step": 20, "gsm8k_n": 256, "gsm8k_quick_chat": 0.625, "t_eval_s": 21.5} +{"step": 21, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6477050055406535, "tokens": 174723, "cumulative_loss_tokens": 3930879, "grad_norm": 1.1796875, "lr": 3e-05, "finish_rate": 0.984, "comp_len": 682.5, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 51.9, "t_step_s": 125.4, "t_refresh_s": 0.3, "mem_gb": 10.81} +{"step": 22, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6475780201495807, "tokens": 180570, "cumulative_loss_tokens": 4111449, "grad_norm": 1.3203125, "lr": 3e-05, "finish_rate": 0.992, "comp_len": 705.4, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 54.0, "t_step_s": 129.4, "t_refresh_s": 0.3, "mem_gb": 10.86} +{"step": 23, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7763283667631196, "tokens": 156909, "cumulative_loss_tokens": 4268358, "grad_norm": 2.859375, "lr": 3e-05, "finish_rate": 0.992, "comp_len": 612.9, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 47.4, "t_step_s": 117.5, "t_refresh_s": 0.3, "mem_gb": 10.79} +{"step": 24, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6585022193530402, "tokens": 159268, "cumulative_loss_tokens": 4427626, "grad_norm": 1.890625, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 622.1, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 47.8, "t_step_s": 118.8, "t_refresh_s": 0.3, "mem_gb": 10.65} +{"step": 25, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6866574399858845, "tokens": 178444, "cumulative_loss_tokens": 4606070, "grad_norm": 2.140625, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 697.0, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 52.7, "t_step_s": 127.0, "t_refresh_s": 0.3, "mem_gb": 10.99} +{"step": 26, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6220901493041023, "tokens": 164396, "cumulative_loss_tokens": 4770466, "grad_norm": 1.890625, "lr": 3e-05, "finish_rate": 0.996, "comp_len": 642.2, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 50.4, "t_step_s": 123.7, "t_refresh_s": 0.3, "mem_gb": 10.65} +{"step": 27, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7198778484349251, "tokens": 160282, "cumulative_loss_tokens": 4930748, "grad_norm": 2.234375, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 626.1, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 48.8, "t_step_s": 120.2, "t_refresh_s": 0.3, "mem_gb": 10.6} +{"step": 28, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6741528202452416, "tokens": 172160, "cumulative_loss_tokens": 5102908, "grad_norm": 1.359375, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 672.5, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 50.7, "t_step_s": 124.1, "t_refresh_s": 0.3, "mem_gb": 10.68} +{"step": 29, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7210347779406289, "tokens": 160068, "cumulative_loss_tokens": 5262976, "grad_norm": 1.28125, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 625.3, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 48.6, "t_step_s": 119.7, "t_refresh_s": 0.3, "mem_gb": 11.08} +{"step": 30, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7013487155993808, "tokens": 183164, "cumulative_loss_tokens": 5446140, "grad_norm": 2.90625, "lr": 3e-05, "finish_rate": 0.945, "comp_len": 715.5, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 54.7, "t_step_s": 134.9, "t_refresh_s": 0.3, "mem_gb": 10.92} +{"step": 31, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7436684049740376, "tokens": 218267, "cumulative_loss_tokens": 5664407, "grad_norm": 5.09375, "lr": 3e-05, "finish_rate": 0.891, "comp_len": 852.6, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 65.5, "t_step_s": 151.0, "t_refresh_s": 0.3, "mem_gb": 11.02} +{"step": 32, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7595977832707371, "tokens": 197139, "cumulative_loss_tokens": 5861546, "grad_norm": 3.703125, "lr": 3e-05, "finish_rate": 0.91, "comp_len": 770.1, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 60.9, "t_step_s": 143.7, "t_refresh_s": 0.3, "mem_gb": 11.05} +{"step": 33, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8273114228759962, "tokens": 216428, "cumulative_loss_tokens": 6077974, "grad_norm": 4.78125, "lr": 3e-05, "finish_rate": 0.848, "comp_len": 845.4, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 66.3, "t_step_s": 151.8, "t_refresh_s": 0.3, "mem_gb": 11.03} +{"step": 34, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6696754721427348, "tokens": 238751, "cumulative_loss_tokens": 6316725, "grad_norm": 2.296875, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 932.6, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 73.3, "t_step_s": 164.8, "t_refresh_s": 0.3, "mem_gb": 11.05} +{"step": 35, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8284414786866234, "tokens": 220947, "cumulative_loss_tokens": 6537672, "grad_norm": 2.734375, "lr": 3e-05, "finish_rate": 0.84, "comp_len": 863.1, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 67.6, "t_step_s": 157.2, "t_refresh_s": 0.3, "mem_gb": 11.11} +{"step": 36, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7958944965435142, "tokens": 242997, "cumulative_loss_tokens": 6780669, "grad_norm": 2.328125, "lr": 3e-05, "finish_rate": 0.816, "comp_len": 949.2, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 74.9, "t_step_s": 166.2, "t_refresh_s": 0.3, "mem_gb": 11.03} +{"step": 37, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8602971678411787, "tokens": 230918, "cumulative_loss_tokens": 7011587, "grad_norm": 4.0625, "lr": 3e-05, "finish_rate": 0.852, "comp_len": 902.0, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 71.9, "t_step_s": 159.4, "t_refresh_s": 0.3, "mem_gb": 10.9} +{"step": 38, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7496087419952795, "tokens": 301447, "cumulative_loss_tokens": 7313034, "grad_norm": 3.28125, "lr": 3e-05, "finish_rate": 0.75, "comp_len": 1177.5, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 110.3, "t_step_s": 211.0, "t_refresh_s": 0.3, "mem_gb": 11.02} +{"step": 39, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7658825440907994, "tokens": 263184, "cumulative_loss_tokens": 7576218, "grad_norm": 3.03125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 1028.1, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 91.7, "t_step_s": 185.8, "t_refresh_s": 0.3, "mem_gb": 10.96} +{"step": 40, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.848072817392145, "tokens": 269682, "cumulative_loss_tokens": 7845900, "grad_norm": 2.828125, "lr": 3e-05, "finish_rate": 0.797, "comp_len": 1053.4, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 93.4, "t_step_s": 186.8, "t_refresh_s": 0.3, "mem_gb": 11.04} +{"step": 40, "gsm8k_n": 256, "gsm8k_quick_chat": 0.54296875, "t_eval_s": 31.7} +{"step": 41, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8468266629074019, "tokens": 269903, "cumulative_loss_tokens": 8115803, "grad_norm": 1.9140625, "lr": 3e-05, "finish_rate": 0.812, "comp_len": 1054.3, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 94.9, "t_step_s": 189.0, "t_refresh_s": 0.3, "mem_gb": 10.93} +{"step": 42, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8907068655281217, "tokens": 236636, "cumulative_loss_tokens": 8352439, "grad_norm": 2.375, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 924.4, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 69.4, "t_step_s": 157.3, "t_refresh_s": 0.3, "mem_gb": 10.96} +{"step": 43, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6942223681318705, "tokens": 259699, "cumulative_loss_tokens": 8612138, "grad_norm": 1.640625, "lr": 3e-05, "finish_rate": 0.832, "comp_len": 1014.4, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 86.2, "t_step_s": 179.7, "t_refresh_s": 0.3, "mem_gb": 11.0} +{"step": 44, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.9662351564620688, "tokens": 218283, "cumulative_loss_tokens": 8830421, "grad_norm": 1.7109375, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 852.7, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 64.8, "t_step_s": 148.7, "t_refresh_s": 0.3, "mem_gb": 10.93} +{"step": 45, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7985041718275216, "tokens": 205156, "cumulative_loss_tokens": 9035577, "grad_norm": 1.359375, "lr": 3e-05, "finish_rate": 0.934, "comp_len": 801.4, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 58.4, "t_step_s": 138.4, "t_refresh_s": 0.3, "mem_gb": 11.02} +{"step": 46, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7015555856755568, "tokens": 195285, "cumulative_loss_tokens": 9230862, "grad_norm": 1.6953125, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 762.8, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 56.2, "t_step_s": 134.2, "t_refresh_s": 0.3, "mem_gb": 10.84} +{"step": 47, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.80149552587993, "tokens": 200589, "cumulative_loss_tokens": 9431451, "grad_norm": 1.453125, "lr": 3e-05, "finish_rate": 0.93, "comp_len": 783.6, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 59.1, "t_step_s": 139.4, "t_refresh_s": 0.3, "mem_gb": 10.9} +{"step": 48, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7283367961467688, "tokens": 161241, "cumulative_loss_tokens": 9592692, "grad_norm": 2.3125, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 629.8, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 48.0, "t_step_s": 117.7, "t_refresh_s": 0.3, "mem_gb": 11.0} +{"step": 49, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7235594729713566, "tokens": 179894, "cumulative_loss_tokens": 9772586, "grad_norm": 2.609375, "lr": 3e-05, "finish_rate": 0.934, "comp_len": 702.7, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 53.3, "t_step_s": 132.3, "t_refresh_s": 0.3, "mem_gb": 10.8} +{"step": 50, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8555238536313134, "tokens": 189930, "cumulative_loss_tokens": 9962516, "grad_norm": 1.7578125, "lr": 3e-05, "finish_rate": 0.934, "comp_len": 741.9, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 56.0, "t_step_s": 135.8, "t_refresh_s": 0.3, "mem_gb": 10.97} +{"step": 51, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.1202008084165977, "tokens": 201080, "cumulative_loss_tokens": 10163596, "grad_norm": 2.296875, "lr": 3e-05, "finish_rate": 0.871, "comp_len": 785.5, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 58.7, "t_step_s": 141.7, "t_refresh_s": 0.3, "mem_gb": 10.94} +{"step": 52, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.0697661634978188, "tokens": 227125, "cumulative_loss_tokens": 10390721, "grad_norm": 7.0, "lr": 3e-05, "finish_rate": 0.797, "comp_len": 887.2, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 69.8, "t_step_s": 157.5, "t_refresh_s": 0.3, "mem_gb": 11.02} +{"step": 53, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.575091722504929, "tokens": 269889, "cumulative_loss_tokens": 10660610, "grad_norm": 42.5, "lr": 3e-05, "finish_rate": 0.715, "comp_len": 1054.3, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 103.5, "t_step_s": 202.4, "t_refresh_s": 0.3, "mem_gb": 11.03} +{"step": 54, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.5818150336665064, "tokens": 270119, "cumulative_loss_tokens": 10930729, "grad_norm": 53.75, "lr": 3e-05, "finish_rate": 0.684, "comp_len": 1055.2, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 100.1, "t_step_s": 199.4, "t_refresh_s": 0.3, "mem_gb": 11.03} +{"step": 55, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.1974753907408464, "tokens": 251568, "cumulative_loss_tokens": 11182297, "grad_norm": 30.0, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 982.7, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 92.6, "t_step_s": 188.2, "t_refresh_s": 0.3, "mem_gb": 11.0} +{"step": 56, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.9958612417439423, "tokens": 226172, "cumulative_loss_tokens": 11408469, "grad_norm": 11.625, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 883.5, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 71.2, "t_step_s": 162.8, "t_refresh_s": 0.3, "mem_gb": 11.04} +{"step": 57, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.1004676886661402, "tokens": 242921, "cumulative_loss_tokens": 11651390, "grad_norm": 7.6875, "lr": 3e-05, "finish_rate": 0.871, "comp_len": 948.9, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 82.8, "t_step_s": 173.6, "t_refresh_s": 0.3, "mem_gb": 10.8} +{"step": 58, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.9067449068144753, "tokens": 248976, "cumulative_loss_tokens": 11900366, "grad_norm": 6.0625, "lr": 3e-05, "finish_rate": 0.895, "comp_len": 972.6, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 81.6, "t_step_s": 171.3, "t_refresh_s": 0.3, "mem_gb": 10.95} +{"step": 59, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.9248105504721407, "tokens": 277195, "cumulative_loss_tokens": 12177561, "grad_norm": 12.0625, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 1082.8, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 96.5, "t_step_s": 192.0, "t_refresh_s": 0.3, "mem_gb": 10.98} +{"step": 60, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.9872396219904398, "tokens": 279039, "cumulative_loss_tokens": 12456600, "grad_norm": 7.34375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 1090.0, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 98.3, "t_step_s": 181.7, "t_refresh_s": 0.0, "mem_gb": 11.02} +{"step": 60, "gsm8k_n": 256, "gsm8k_quick_chat": 0.52734375, "t_eval_s": 33.5} diff --git a/healed/refkl_keep50_s1223/vllm_server.log b/healed/refkl_keep50_s1223/vllm_server.log new file mode 100644 index 0000000000000000000000000000000000000000..88ca4a1112a2616f459abf96201d78f60622de49 --- /dev/null +++ b/healed/refkl_keep50_s1223/vllm_server.log @@ -0,0 +1,4460 @@ +Skipping import of cpp extensions due to incompatible torch version 2.10.0+cu128 for torchao version 0.15.0 Please see https://github.com/pytorch/ao/issues/2919 for more info +WARNING 07-30 09:23:04 [registry.py:915] Model architecture OlmoeForCausalLM is already registered, and will be overwritten by the new model class glean_vllm.pruned_olmoe:PrunedOlmoeForCausalLM. +(APIServer pid=1960886) INFO 07-30 09:23:04 [utils.py:299] +(APIServer pid=1960886) INFO 07-30 09:23:04 [utils.py:299] █ █ █▄ ▄█ +(APIServer pid=1960886) INFO 07-30 09:23:04 [utils.py:299] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.19.0 +(APIServer pid=1960886) INFO 07-30 09:23:04 [utils.py:299] █▄█▀ █ █ █ █ model outputs/pruned/glean-0125inst-math-keep50 +(APIServer pid=1960886) INFO 07-30 09:23:04 [utils.py:299] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀ +(APIServer pid=1960886) INFO 07-30 09:23:04 [utils.py:299] +(APIServer pid=1960886) INFO 07-30 09:23:04 [utils.py:233] non-default args: {'model_tag': 'outputs/pruned/glean-0125inst-math-keep50', 'host': '127.0.0.1', 'port': 8377, 'model': 'outputs/pruned/glean-0125inst-math-keep50', 'max_model_len': 3200, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85} +(APIServer pid=1960886) INFO 07-30 09:23:12 [model.py:549] Resolved architecture: OlmoeForCausalLM +(APIServer pid=1960886) INFO 07-30 09:23:12 [model.py:1678] Using max model len 3200 +(APIServer pid=1960886) INFO 07-30 09:23:12 [vllm.py:790] Asynchronous scheduling is enabled. +(APIServer pid=1960886) WARNING 07-30 09:23:12 [vllm.py:848] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none +(APIServer pid=1960886) WARNING 07-30 09:23:12 [vllm.py:859] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored. +(APIServer pid=1960886) INFO 07-30 09:23:12 [vllm.py:1025] Cudagraph is disabled under eager mode +(APIServer pid=1960886) INFO 07-30 09:23:12 [compilation.py:290] Enabled custom fusions: norm_quant, act_quant +Skipping import of cpp extensions due to incompatible torch version 2.10.0+cu128 for torchao version 0.15.0 Please see https://github.com/pytorch/ao/issues/2919 for more info +(EngineCore pid=1961198) WARNING 07-30 09:23:21 [registry.py:915] Model architecture OlmoeForCausalLM is already registered, and will be overwritten by the new model class glean_vllm.pruned_olmoe:PrunedOlmoeForCausalLM. +(EngineCore pid=1961198) INFO 07-30 09:23:21 [core.py:105] Initializing a V1 LLM engine (v0.19.0) with config: model='outputs/pruned/glean-0125inst-math-keep50', speculative_config=None, tokenizer='outputs/pruned/glean-0125inst-math-keep50', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=3200, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=1961198) INFO 07-30 09:23:21 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:35299 backend=nccl +(EngineCore pid=1961198) INFO 07-30 09:23:21 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=1961198) INFO 07-30 09:23:21 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep50... +(EngineCore pid=1961198) INFO 07-30 09:23:22 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=1961198) INFO 07-30 09:23:22 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=1961198) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=1954237) INFO 07-30 08:17:17 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:35307 backend=nccl +(EngineCore pid=1954237) INFO 07-30 08:17:17 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=1954237) INFO 07-30 08:17:18 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep50... +(EngineCore pid=1954237) INFO 07-30 08:17:18 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=1954237) INFO 07-30 08:17:18 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=1954237) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=89849) INFO 08-01 18:36:05 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.27:48265 backend=nccl +(EngineCore pid=89849) INFO 08-01 18:36:05 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=89849) INFO 08-01 18:36:06 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep50... +(EngineCore pid=89849) INFO 08-01 18:36:06 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=89849) INFO 08-01 18:36:06 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=89849) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=2031228) INFO 07-30 22:24:31 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:42255 backend=nccl +(EngineCore pid=2031228) INFO 07-30 22:24:31 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=2031228) INFO 07-30 22:24:32 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep50... +(EngineCore pid=2031228) INFO 07-30 22:24:32 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=2031228) INFO 07-30 22:24:32 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=2031228) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=296300) INFO 07-12 08:11:08 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:59759 backend=nccl +(EngineCore pid=296300) INFO 07-12 08:11:08 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=296300) INFO 07-12 08:11:08 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep5... +(EngineCore pid=296300) INFO 07-12 08:11:09 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=296300) INFO 07-12 08:11:09 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=296300) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=298456) INFO 07-12 08:21:28 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:55783 backend=nccl +(EngineCore pid=298456) INFO 07-12 08:21:28 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=298456) INFO 07-12 08:21:28 [gpu_model_runner.py:4735] Starting to load model outputs/healed/sweep_vllm_3e5/vllm_live... +(EngineCore pid=298456) INFO 07-12 08:21:29 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=298456) INFO 07-12 08:21:29 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=298456) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=300027) INFO 07-12 08:23:53 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:47897 backend=nccl +(EngineCore pid=300027) INFO 07-12 08:23:53 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=300027) INFO 07-12 08:23:54 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep5... +(EngineCore pid=300027) INFO 07-12 08:23:54 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=300027) INFO 07-12 08:23:54 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=300027) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=301850) INFO 07-12 08:34:14 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:51247 backend=nccl +(EngineCore pid=301850) INFO 07-12 08:34:14 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=301850) INFO 07-12 08:34:14 [gpu_model_runner.py:4735] Starting to load model outputs/healed/sweep_vllm_3e5/vllm_live... +(EngineCore pid=301850) INFO 07-12 08:34:15 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=301850) INFO 07-12 08:34:15 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=301850) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=303276) INFO 07-12 08:43:09 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:34477 backend=nccl +(EngineCore pid=303276) INFO 07-12 08:43:09 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=303276) INFO 07-12 08:43:10 [gpu_model_runner.py:4735] Starting to load model outputs/healed/sweep_vllm_3e5/vllm_live... +(EngineCore pid=303276) INFO 07-12 08:43:11 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=303276) INFO 07-12 08:43:11 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=303276) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []} +(EngineCore pid=304870) INFO 07-12 08:52:43 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:40663 backend=nccl +(EngineCore pid=304870) INFO 07-12 08:52:43 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=304870) INFO 07-12 08:52:43 [gpu_model_runner.py:4735] Starting to load model outputs/healed/sweep_vllm_3e5/vllm_live... +(EngineCore pid=304870) INFO 07-12 08:52:44 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. +(EngineCore pid=304870) INFO 07-12 08:52:44 [flash_attn.py:596] Using FlashAttention version 2 +(EngineCore pid=304870) Loading safetensors checkpoint shards: 0% Completed | 0/2 [00:00