Add files using upload-large-folder tool
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- evals/calib_seed_overlap_0125inst_c4.json +28 -0
- evals/general_suite_smoke.log +26 -0
- evals/protocolF_finale.log +0 -0
- evals/protocolF_queue.log +0 -0
- evals/protocol_c_backlog.log +35 -0
- healed/correctness_ab/glean_keep25_nogold_s1224.console.log +233 -0
- healed/correctness_ab/glean_keep25_nogold_s1225.console.log +232 -0
- healed/correctness_ab/glean_keep25_nogold_s1226.console.log +187 -0
- healed/grid_math/glean_keep25_s1224.console.log +231 -0
- healed/grid_math/glean_keep75_s1224.console.log +232 -0
- healed/grid_math/glean_keep75_s1225.console.log +120 -0
- healed/grid_math/glean_keep75_s1226.console.log +120 -0
- healed/grid_math/grid.log +0 -0
- healed/grid_math/reap_keep25_s1225.console.log +231 -0
- healed/grid_math/reap_keep75_s1225.console.log +231 -0
- healed/grid_math/reap_keep75_s1226.console.log +284 -0
- healed/grid_math/uniform_keep50_s1226.console.log +232 -0
- healed/grid_math/worker_s1224.log +22 -0
- healed/grid_math/worker_s1225.log +22 -0
- healed/keep50_offpolicy_warmup_s1224/args.json +71 -0
- healed/keep50_offpolicy_warmup_s1224/train_log.jsonl +150 -0
- healed/knee0924/keep20.console.log +18 -0
- healed/knee0924/keep25.console.log +18 -0
- healed/knee0924/keep30.console.log +18 -0
- healed/knee0924/keep40.console.log +19 -0
- healed/knee0924/keep50.console.log +19 -0
- healed/mixdistill_smoke/args.json +69 -0
- healed/mixdistill_smoke/train_log.jsonl +6 -0
- healed/mixdistill_smoke/vllm_server.log +446 -0
- healed/soak2048_filtered_keep50_s1223/args.json +64 -0
- healed/soak2048_filtered_keep50_s1223/train_log.jsonl +105 -0
- healed/soak2048_filtered_keep50_s1223/vllm_server.log +0 -0
- healed/soak2048_filtered_keep50_s1223/wandb_sync.log +2 -0
- healed/stableopd_cold_keep50/args.json +71 -0
- healed/stableopd_cold_keep50/train_log.jsonl +127 -0
- healed/stableopd_cold_keep50/vllm_server.log +0 -0
- healed/stableopd_cold_keep50/wandb_sync.log +2 -0
- pruned/glean-0125inst-math-keep50/chat_template.jinja +9 -0
- pruned/glean-0125inst-math-keep50/config.json +887 -0
- pruned/glean-0125inst-math-keep50/configuration_pruned_olmoe.py +31 -0
- pruned/glean-0125inst-math-keep50/generation_config.json +6 -0
- pruned/glean-0125inst-math-keep50/model.safetensors.index.json +0 -0
- pruned/glean-0125inst-math-keep50/modeling_pruned_olmoe.py +66 -0
- pruned/glean-0125inst-math-keep50/special_tokens_map.json +23 -0
- pruned/glean-0125inst-math-keep50/tokenizer.json +0 -0
- pruned/glean-0125inst-math-keep50/tokenizer_config.json +247 -0
- pruned/knee0924/keep20.log +2 -0
- pruned/knee0924/keep25.log +2 -0
- pruned/knee0924/keep30.log +2 -0
- pruned/knee0924/keep50.log +2 -0
evals/calib_seed_overlap_0125inst_c4.json
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"stats_a": "outputs/stats_0125inst_c4_0.5M.pt",
|
| 3 |
+
"stats_b": "outputs/stats_0125inst_c4_0.5M_seed2.pt",
|
| 4 |
+
"criterion": "reap",
|
| 5 |
+
"keeps": {
|
| 6 |
+
"0.25": {
|
| 7 |
+
"jaccard_kept_channels": 0.8658533547337816,
|
| 8 |
+
"width_pearson": 0.9878049492835999,
|
| 9 |
+
"width_spearman": 0.987792432308197,
|
| 10 |
+
"max_width_delta": 525,
|
| 11 |
+
"mean_width_delta": 17.361328125
|
| 12 |
+
},
|
| 13 |
+
"0.50": {
|
| 14 |
+
"jaccard_kept_channels": 0.8989887282315784,
|
| 15 |
+
"width_pearson": 0.9867013096809387,
|
| 16 |
+
"width_spearman": 0.9869402647018433,
|
| 17 |
+
"max_width_delta": 509,
|
| 18 |
+
"mean_width_delta": 27.728515625
|
| 19 |
+
},
|
| 20 |
+
"0.75": {
|
| 21 |
+
"jaccard_kept_channels": 0.9456001266666749,
|
| 22 |
+
"width_pearson": 0.9852995872497559,
|
| 23 |
+
"width_spearman": 0.9842259883880615,
|
| 24 |
+
"max_width_delta": 574,
|
| 25 |
+
"mean_width_delta": 25.935546875
|
| 26 |
+
}
|
| 27 |
+
}
|
| 28 |
+
}
|
evals/general_suite_smoke.log
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/20 [00:00<?, ?it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T10:11:28-07:00 serving allenai/OLMoE-1B-7B-0125-Instruct on GPU 0 port 8399
|
| 2 |
+
2026-07-17T10:11:28-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T10:11:58-07:00 server up; running lm_eval [gsm8k]
|
| 4 |
+
2026-07-17:10:11:59 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 5 |
+
2026-07-17:10:12:06 INFO [_cli.run:388] Selected Tasks: ['gsm8k']
|
| 6 |
+
2026-07-17:10:12:07 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 7 |
+
2026-07-17:10:12:07 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}
|
| 8 |
+
2026-07-17:10:12:07 INFO [models.api_models:179] Using max length 2048 - 1
|
| 9 |
+
2026-07-17:10:12:07 INFO [models.api_models:200] Using tokenizer None
|
| 10 |
+
2026-07-17:10:12:08 INFO [evaluator_utils:446] Selected tasks:
|
| 11 |
+
2026-07-17:10:12:08 INFO [evaluator_utils:480] Task: gsm8k (gsm8k/gsm8k.yaml)
|
| 12 |
+
2026-07-17:10:12:08 INFO [evaluator:314] gsm8k: Using gen_kwargs: {'until': ['Question:', '</s>', '<|im_end|>'], 'do_sample': False, 'temperature': 0.0}
|
| 13 |
+
2026-07-17:10:12:08 INFO [api.task:312] Building contexts for gsm8k on rank 0...
|
| 14 |
+
|
| 15 |
0%| | 0/20 [00:00<?, ?it/s]
|
| 16 |
+
2026-07-17:10:12:09 INFO [evaluator:585] Running generate_until requests
|
| 17 |
+
2026-07-17:10:12:09 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 18 |
+
|
| 19 |
+
2026-07-17:10:12:15 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 20 |
+
2026-07-17:10:12:15 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/smoke_base/student/*.jsonl
|
| 21 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: 20.0, num_fewshot: None, batch_size: 1
|
| 22 |
+
|Tasks|Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 23 |
+
|-----|------:|----------------|-----:|-----------|---|----:|---|-----:|
|
| 24 |
+
|gsm8k| 3|flexible-extract| 5|exact_match|↑ | 0.55|± |0.1141|
|
| 25 |
+
| | |strict-match | 5|exact_match|↑ | 0.55|± |0.1141|
|
| 26 |
+
|
| 27 |
+
2026-07-17T10:12:16-07:00 lm_eval exit=0 -> outputs/evals/general_suite/smoke_base
|
evals/protocolF_finale.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/protocolF_queue.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/protocol_c_backlog.log
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-20T08:37:27-07:00 stage 1: collecting seed-2 calibration stats
|
| 2 |
+
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
tokens seen: 499,712 (244 seqs x 2048, dataset=c4)
|
| 6 |
+
token_count [L=16, E=64] — routed-token counts (top-k membership):
|
| 7 |
+
per-layer row sums (== n_tokens * top_k): min 3997696, max 3997696
|
| 8 |
+
per-expert counts: min 10984, median 61647, max 270930, mean 62464.0
|
| 9 |
+
zero-routed (layer, expert) pairs: 0 / 1024
|
| 10 |
+
saved -> outputs/stats_0125inst_c4_0.5M_seed2.pt
|
| 11 |
+
2026-07-20T08:39:14-07:00 stage 2: policy-pair Protocol-C evals
|
| 12 |
+
2026-07-20T08:39:14-07:00 EVAL off_forward_seed1224 on GPU 0 (port 8430)
|
| 13 |
+
2026-07-20T08:41:44-07:00 EVAL off_forward_seed1225 on GPU 1 (port 8431)
|
| 14 |
+
2026-07-20T08:44:14-07:00 EVAL off_forward_seed1226 on GPU 2 (port 8432)
|
| 15 |
+
2026-07-20T08:54:10-07:00 EVAL on_reverse_seed1224 on GPU 0 (port 8430)
|
| 16 |
+
2026-07-20T08:57:11-07:00 EVAL on_reverse_seed1225 on GPU 1 (port 8431)
|
| 17 |
+
2026-07-20T08:59:04-07:00 EVAL on_reverse_seed1226 on GPU 2 (port 8432)
|
| 18 |
+
2026-07-20T09:10:10-07:00 EVAL combo_on25_seed1224 on GPU 0 (port 8430)
|
| 19 |
+
2026-07-20T09:14:00-07:00 EVAL combo_on25_seed1225 on GPU 1 (port 8431)
|
| 20 |
+
2026-07-20T09:15:51-07:00 EVAL combo_on25_seed1226 on GPU 2 (port 8432)
|
| 21 |
+
2026-07-20T09:25:56-07:00 lane GPU0 complete
|
| 22 |
+
2026-07-20T09:30:21-07:00 lane GPU1 complete
|
| 23 |
+
2026-07-20T09:31:33-07:00 lane GPU2 complete
|
| 24 |
+
2026-07-20T09:31:33-07:00 stage 3: allocation overlap report
|
| 25 |
+
Traceback (most recent call last):
|
| 26 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/scripts/43_alloc_overlap.py", line 88, in <module>
|
| 27 |
+
main()
|
| 28 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/scripts/43_alloc_overlap.py", line 61, in main
|
| 29 |
+
sa = layer_scores(stats_a, args.criterion)
|
| 30 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 31 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/scripts/43_alloc_overlap.py", line 31, in layer_scores
|
| 32 |
+
return channel_scores(stats, b=b, alpha=alpha, beta=beta) # [L, E, C]
|
| 33 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 34 |
+
TypeError: channel_scores() missing 1 required positional argument: 'colnorms'
|
| 35 |
+
2026-07-20T09:31:34-07:00 ###### PROTOCOL-C BACKLOG COMPLETE ######
|
healed/correctness_ab/glean_keep25_nogold_s1224.console.log
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: setting up run vf58ltja
|
| 6 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 7 |
+
wandb: Run data is saved locally in outputs/healed/correctness_ab/glean_keep25_nogold_s1224/wandb/run-20260717_042106-vf58ltja
|
| 8 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 9 |
+
wandb: Syncing run glean-nogold-keep25-s1224
|
| 10 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 11 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/vf58ltja
|
| 12 |
+
9918 cached top-128 chat trajectories / 6,476,712 unique tokens | 53 steps/epoch | 150 total steps | student params 2.09B | teacher overlap=False
|
| 13 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.343373699468871, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 92.0, "lr": 6e-06, "finish_rate": 0.357, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 182}, "mem_gb": 10.05}
|
| 14 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 15 |
+
[eval step 1] sample: '\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###'
|
| 16 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.3397573178668818, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 87.0, "lr": 9e-06, "finish_rate": 0.406, "comp_len": 641.7, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 187}, "mem_gb": 10.02}
|
| 17 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.0775470499475797, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 57.5, "lr": 1.2e-05, "finish_rate": 0.49, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 200}, "mem_gb": 10.01}
|
| 18 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8723263926585515, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 13.1875, "lr": 1.5e-05, "finish_rate": 0.482, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 193}, "mem_gb": 10.09}
|
| 19 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.808311350060006, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 11.1875, "lr": 1.8e-05, "finish_rate": 0.418, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 182}, "mem_gb": 10.03}
|
| 20 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6887235863516729, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 5.78125, "lr": 2.1e-05, "finish_rate": 0.425, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 181}, "mem_gb": 10.02}
|
| 21 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5664163419249157, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 3.734375, "lr": 2.4e-05, "finish_rate": 0.503, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 193}, "mem_gb": 10.01}
|
| 22 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4520683725203077, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 2.234375, "lr": 2.7000000000000002e-05, "finish_rate": 0.495, "comp_len": 625.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 192}, "mem_gb": 10.03}
|
| 23 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5079976932493349, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 2.515625, "lr": 3e-05, "finish_rate": 0.446, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 186}, "mem_gb": 10.06}
|
| 24 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3765195841965576, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 1.453125, "lr": 3e-05, "finish_rate": 0.394, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 180}, "mem_gb": 10.02}
|
| 25 |
+
[eval step 10] sample: "To solve this problem, we need to determine the coordinates of the center of the sphere given the conditions and then find the coordinates of the center of the sphere shifted by \\((1,1,1)\\).\n\nLet's br"
|
| 26 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32854144711097083, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 1.2734375, "lr": 3e-05, "finish_rate": 0.526, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.6, "frames": {"chat": 196}, "mem_gb": 10.01}
|
| 27 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32865148959445456, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 1.0546875, "lr": 3e-05, "finish_rate": 0.36, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 178}, "mem_gb": 10.04}
|
| 28 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3404995323523879, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 1.0703125, "lr": 3e-05, "finish_rate": 0.294, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 177}, "mem_gb": 10.12}
|
| 29 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27378314006154736, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.76953125, "lr": 3e-05, "finish_rate": 0.424, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 184}, "mem_gb": 10.04}
|
| 30 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.31704738569681845, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 2.171875, "lr": 3e-05, "finish_rate": 0.36, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 186}, "mem_gb": 10.04}
|
| 31 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2564430236879736, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.6640625, "lr": 3e-05, "finish_rate": 0.363, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 182}, "mem_gb": 10.02}
|
| 32 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2718902955224117, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.76953125, "lr": 3e-05, "finish_rate": 0.296, "comp_len": 670.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 179}, "mem_gb": 10.11}
|
| 33 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25489308905216557, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.6953125, "lr": 3e-05, "finish_rate": 0.513, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 193}, "mem_gb": 10.11}
|
| 34 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27514486570470037, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.74609375, "lr": 3e-05, "finish_rate": 0.316, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 174}, "mem_gb": 10.14}
|
| 35 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21634715970903634, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.60546875, "lr": 3e-05, "finish_rate": 0.464, "comp_len": 618.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 194}, "mem_gb": 10.01}
|
| 36 |
+
[eval step 20] sample: 'To solve this problem, we need to determine the coordinates of the center \\((p, q, r)\\) of the sphere that passes through the origin \\((0, 0, 0)\\) and intersects the \\(x\\)-axis, \\(y\\)-axis, and \\(z\\)-'
|
| 37 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24452172587576013, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.4, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.6, "frames": {"chat": 180}, "mem_gb": 10.1}
|
| 38 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23485065581947565, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.65234375, "lr": 3e-05, "finish_rate": 0.53, "comp_len": 606.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 198}, "mem_gb": 10.01}
|
| 39 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24486064562636117, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.354, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 178}, "mem_gb": 10.15}
|
| 40 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2208150434208413, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.65625, "lr": 3e-05, "finish_rate": 0.484, "comp_len": 631.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 190}, "mem_gb": 10.01}
|
| 41 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20461380687194566, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.54296875, "lr": 3e-05, "finish_rate": 0.345, "comp_len": 701.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 171}, "mem_gb": 10.05}
|
| 42 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21616830685002109, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.74609375, "lr": 3e-05, "finish_rate": 0.316, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 174}, "mem_gb": 10.16}
|
| 43 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22327683328259737, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.442, "comp_len": 631.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 190}, "mem_gb": 10.08}
|
| 44 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2106279812599222, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.5546875, "lr": 3e-05, "finish_rate": 0.253, "comp_len": 705.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 170}, "mem_gb": 10.12}
|
| 45 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21575977770760654, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.55078125, "lr": 3e-05, "finish_rate": 0.347, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 176}, "mem_gb": 10.1}
|
| 46 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24658188070257506, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.6953125, "lr": 3e-05, "finish_rate": 0.452, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.6, "frames": {"chat": 188}, "mem_gb": 10.01}
|
| 47 |
+
[eval step 30] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - We have a plane passing through the origin \\((0,0,0)\\) and intersecting the \\(x\\)-axis, \\(y\\)-axis, and \\(z\\'
|
| 48 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19494453672828774, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.55, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 202}, "mem_gb": 10.02}
|
| 49 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 50 |
+
warnings.warn('Grouped GEMM not available.')
|
| 51 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 52 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 53 |
+
wandb: setting up run k963lqxn
|
| 54 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 55 |
+
wandb: Run data is saved locally in outputs/healed/correctness_ab/glean_keep25_nogold_s1224/wandb/run-20260717_043837-k963lqxn
|
| 56 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 57 |
+
wandb: Syncing run glean-nogold-keep25-s1224
|
| 58 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 59 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/k963lqxn
|
| 60 |
+
9918 cached top-128 chat trajectories / 6,476,712 unique tokens | 53 steps/epoch | 100 total steps | student params 2.09B | teacher overlap=False
|
| 61 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.343373699468871, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 92.0, "lr": 6e-06, "finish_rate": 0.357, "comp_len": 659.3, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 182}, "mem_gb": 10.05}
|
| 62 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 63 |
+
[eval step 1] sample: '\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###'
|
| 64 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.341547059782346, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 86.5, "lr": 9e-06, "finish_rate": 0.406, "comp_len": 641.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 187}, "mem_gb": 10.02}
|
| 65 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.0767730478396018, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 56.5, "lr": 1.2e-05, "finish_rate": 0.49, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 200}, "mem_gb": 10.01}
|
| 66 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8719748466918866, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 13.125, "lr": 1.5e-05, "finish_rate": 0.482, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 193}, "mem_gb": 10.09}
|
| 67 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8081670430993041, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 11.6875, "lr": 1.8e-05, "finish_rate": 0.418, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 182}, "mem_gb": 10.03}
|
| 68 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 69 |
+
warnings.warn('Grouped GEMM not available.')
|
| 70 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 71 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 72 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 73 |
+
wandb: Run data is saved locally in outputs/healed/correctness_ab/glean_keep25_nogold_s1224/wandb/run-20260717_044202-rgbwy6bs
|
| 74 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 75 |
+
wandb: Syncing run glean-nogold-keep25-s1224
|
| 76 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 77 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/rgbwy6bs
|
| 78 |
+
9918 cached top-128 chat trajectories / 6,476,712 unique tokens | 53 steps/epoch | 100 total steps | student params 2.09B | teacher overlap=False
|
| 79 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.343373699468871, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 92.0, "lr": 6e-06, "finish_rate": 0.357, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 182}, "mem_gb": 10.05}
|
| 80 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 81 |
+
[eval step 1] sample: '\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###'
|
| 82 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.3382042271584271, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 87.0, "lr": 9e-06, "finish_rate": 0.406, "comp_len": 641.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 187}, "mem_gb": 10.02}
|
| 83 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.0746954432512323, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 56.25, "lr": 1.2e-05, "finish_rate": 0.49, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 200}, "mem_gb": 10.01}
|
| 84 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8724394266794125, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 13.0625, "lr": 1.5e-05, "finish_rate": 0.482, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 193}, "mem_gb": 10.09}
|
| 85 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8081986944486698, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 10.5, "lr": 1.8e-05, "finish_rate": 0.418, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 182}, "mem_gb": 10.03}
|
| 86 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6902307474392155, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 6.65625, "lr": 2.1e-05, "finish_rate": 0.425, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 181}, "mem_gb": 10.02}
|
| 87 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5614831128805876, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 3.984375, "lr": 2.4e-05, "finish_rate": 0.503, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 193}, "mem_gb": 10.01}
|
| 88 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4556331973321736, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 2.703125, "lr": 2.7000000000000002e-05, "finish_rate": 0.495, "comp_len": 625.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 192}, "mem_gb": 10.03}
|
| 89 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5146083116727571, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 3.0625, "lr": 3e-05, "finish_rate": 0.446, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 186}, "mem_gb": 10.06}
|
| 90 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3786136556045463, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 1.4609375, "lr": 3e-05, "finish_rate": 0.394, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 180}, "mem_gb": 10.02}
|
| 91 |
+
[eval step 10] sample: 'To solve this problem, we need to determine the coordinates of the center of the sphere given the origin \\((0,0,0)\\), and the points \\(A(a,0,0)\\), \\(B(a,b,0)\\), and \\(C(a,b,0)\\) on'
|
| 92 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32958129375105105, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 1.1796875, "lr": 3e-05, "finish_rate": 0.526, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 196}, "mem_gb": 10.01}
|
| 93 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3299191950291395, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 1.296875, "lr": 3e-05, "finish_rate": 0.36, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 178}, "mem_gb": 10.04}
|
| 94 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.34145802231493094, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.98828125, "lr": 3e-05, "finish_rate": 0.294, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 177}, "mem_gb": 10.12}
|
| 95 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2737439000794043, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.8828125, "lr": 3e-05, "finish_rate": 0.424, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 184}, "mem_gb": 10.04}
|
| 96 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3136025007430464, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 1.8203125, "lr": 3e-05, "finish_rate": 0.36, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 186}, "mem_gb": 10.04}
|
| 97 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2564839167782416, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.69921875, "lr": 3e-05, "finish_rate": 0.363, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 182}, "mem_gb": 10.02}
|
| 98 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27114790187949933, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.7578125, "lr": 3e-05, "finish_rate": 0.296, "comp_len": 670.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 179}, "mem_gb": 10.11}
|
| 99 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2547183968774974, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.703125, "lr": 3e-05, "finish_rate": 0.513, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 193}, "mem_gb": 10.11}
|
| 100 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27456971705394484, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.7265625, "lr": 3e-05, "finish_rate": 0.316, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 174}, "mem_gb": 10.14}
|
| 101 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2159317757766694, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.464, "comp_len": 618.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 194}, "mem_gb": 10.01}
|
| 102 |
+
[eval step 20] sample: 'To solve this problem, we need to determine the coordinates of the center of the sphere and then calculate the required expression \\(\\frac{a}{p} + \\frac{b}{q} + \\frac{c}{r}\\).\n\n### Step-by-Step Soluti'
|
| 103 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2439304939982171, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.4, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 180}, "mem_gb": 10.1}
|
| 104 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23473809580827754, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.73828125, "lr": 3e-05, "finish_rate": 0.53, "comp_len": 606.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 198}, "mem_gb": 10.01}
|
| 105 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24481348447948695, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.66015625, "lr": 3e-05, "finish_rate": 0.354, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 178}, "mem_gb": 10.15}
|
| 106 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21953086602886518, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.64453125, "lr": 3e-05, "finish_rate": 0.484, "comp_len": 631.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 190}, "mem_gb": 10.01}
|
| 107 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20367557740602643, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.345, "comp_len": 701.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 171}, "mem_gb": 10.05}
|
| 108 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21455700743719935, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.316, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 174}, "mem_gb": 10.16}
|
| 109 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2221487521355351, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.442, "comp_len": 631.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 190}, "mem_gb": 10.08}
|
| 110 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2093166130521645, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.253, "comp_len": 705.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 170}, "mem_gb": 10.12}
|
| 111 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21438092577407758, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.347, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 176}, "mem_gb": 10.1}
|
| 112 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24713442737646402, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.7421875, "lr": 3e-05, "finish_rate": 0.452, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 188}, "mem_gb": 10.01}
|
| 113 |
+
[eval step 30] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - We have a plane passing through the origin \\((0,0,0)\\) and intersecting the \\(x\\)-axis, \\(y\\)-axis, and \\(z\\'
|
| 114 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19447750183778503, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.5625, "lr": 3e-05, "finish_rate": 0.55, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 202}, "mem_gb": 10.02}
|
| 115 |
+
{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18556029006137204, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.587, "comp_len": 597.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 201}, "mem_gb": 9.99}
|
| 116 |
+
{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16653565687468896, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.4921875, "lr": 3e-05, "finish_rate": 0.371, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 186}, "mem_gb": 10.01}
|
| 117 |
+
{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.195527940724728, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.322, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 174}, "mem_gb": 10.03}
|
| 118 |
+
{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19189287858692308, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.405, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 185}, "mem_gb": 10.01}
|
| 119 |
+
{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19142657942896088, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.548, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 197}, "mem_gb": 10.07}
|
| 120 |
+
{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1833858589090407, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.299, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 174}, "mem_gb": 10.03}
|
| 121 |
+
{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16718798512797803, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.598, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 209}, "mem_gb": 10.09}
|
| 122 |
+
{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1975579697556173, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.436, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 188}, "mem_gb": 10.02}
|
| 123 |
+
{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19122942256492872, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.379, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 182}, "mem_gb": 10.14}
|
| 124 |
+
[eval step 40] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - The plane passes through the origin \\(O(0,0,0)\\) and intersects the \\(x\\)-axis, \\(y\\)-axis, and \\(z\\)-axis a'
|
| 125 |
+
{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2368633280776441, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.84375, "lr": 3e-05, "finish_rate": 0.28, "comp_len": 685.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 175}, "mem_gb": 10.11}
|
| 126 |
+
{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16281984809990974, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.48, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 196}, "mem_gb": 10.07}
|
| 127 |
+
{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20633705335470537, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.36, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 178}, "mem_gb": 10.09}
|
| 128 |
+
{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1594805426252385, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.397, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 184}, "mem_gb": 10.01}
|
| 129 |
+
{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17871330462048451, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.335, "comp_len": 693.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 173}, "mem_gb": 10.02}
|
| 130 |
+
{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1685610391488299, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.54, "comp_len": 606.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 198}, "mem_gb": 10.19}
|
| 131 |
+
{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17577952838235844, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.291, "comp_len": 697.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.6, "frames": {"chat": 172}, "mem_gb": 10.05}
|
| 132 |
+
{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16250701399346193, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.528, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 193}, "mem_gb": 10.03}
|
| 133 |
+
{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1761657753713429, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.289, "comp_len": 693.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.3, "frames": {"chat": 173}, "mem_gb": 10.04}
|
| 134 |
+
{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16541012109226236, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.198, "comp_len": 718.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 167}, "mem_gb": 10.12}
|
| 135 |
+
[eval step 50] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - We have a plane passing through the origin \\(O(0,0,0)\\) and intersecting the \\(x\\)-axis, \\(y\\)-axis, and \\(z'
|
| 136 |
+
checkpoint snapshot queued -> outputs/healed/correctness_ab/glean_keep25_nogold_s1224/step0050
|
| 137 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17332345672125618, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.256, "comp_len": 714.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 168}, "mem_gb": 10.09}
|
| 138 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1540899394488583, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.415, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 183}, "mem_gb": 10.05}
|
| 139 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15484186229171853, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.446, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 186}, "mem_gb": 10.06}
|
| 140 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13252162307662269, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.266, "comp_len": 693.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 173}, "mem_gb": 10.12}
|
| 141 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13726483113157253, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.558, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 206}, "mem_gb": 10.01}
|
| 142 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12926260091432681, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.523, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 193}, "mem_gb": 10.01}
|
| 143 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12456577034046253, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.312, "comp_len": 693.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 173}, "mem_gb": 10.16}
|
| 144 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12543534933999181, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.452, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 188}, "mem_gb": 10.02}
|
| 145 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11021080381640544, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.357, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 182}, "mem_gb": 10.02}
|
| 146 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12409616308361292, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.578, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 204}, "mem_gb": 10.01}
|
| 147 |
+
[eval step 60] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - We have a plane passing through the origin \\(O(0,0,0)\\) and intersecting the \\(x\\)-axis, \\(y\\)-axis, and \\(z'
|
| 148 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13314346364295732, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.473, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 184}, "mem_gb": 10.02}
|
| 149 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11752055017203093, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.392, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 181}, "mem_gb": 10.13}
|
| 150 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11928972764126956, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.51, "comp_len": 606.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 198}, "mem_gb": 10.06}
|
| 151 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14822008464982112, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.505, "comp_len": 625.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 192}, "mem_gb": 10.09}
|
| 152 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14190797536354513, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.386, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 184}, "mem_gb": 10.01}
|
| 153 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10917766641958927, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.603, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 209}, "mem_gb": 10.01}
|
| 154 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12423299193761, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.251, "comp_len": 701.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 171}, "mem_gb": 10.12}
|
| 155 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12684058725250263, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.384, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 185}, "mem_gb": 10.02}
|
| 156 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1136780122870269, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.378, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 180}, "mem_gb": 10.01}
|
| 157 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14678863985246668, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.436, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 188}, "mem_gb": 10.04}
|
| 158 |
+
[eval step 70] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - The plane passes through the origin \\(O(0,0,0)\\) and intersects the \\(x\\)-axis, \\(y\\)-axis, and \\(z\\)-axis a'
|
| 159 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14580090595372022, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.482, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 195}, "mem_gb": 10.06}
|
| 160 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1124800471910586, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.562, "comp_len": 597.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 201}, "mem_gb": 10.01}
|
| 161 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10710394977213195, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.418, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 177}, "mem_gb": 10.02}
|
| 162 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11889559170119464, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.444, "comp_len": 641.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 187}, "mem_gb": 10.02}
|
| 163 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1331854470112982, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.421, "comp_len": 631.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 190}, "mem_gb": 10.02}
|
| 164 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12192948532433559, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.528, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 197}, "mem_gb": 10.02}
|
| 165 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12339504165450732, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.503, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 193}, "mem_gb": 10.02}
|
| 166 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1338913071037891, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.291, "comp_len": 697.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 172}, "mem_gb": 10.14}
|
| 167 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.118984298597835, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.328, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 174}, "mem_gb": 10.08}
|
| 168 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12922871268056332, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.423, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 182}, "mem_gb": 10.1}
|
| 169 |
+
[eval step 80] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - We have a plane passing through the origin \\(O(0,0,0)\\) and intersecting the \\(x\\)-axis, \\(y\\)-axis, and \\(z'
|
| 170 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1065688143297409, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.419, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 186}, "mem_gb": 10.02}
|
| 171 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10022528152096395, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.415, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 188}, "mem_gb": 10.01}
|
| 172 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11062422166466714, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.413, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 184}, "mem_gb": 10.03}
|
| 173 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13140708254644026, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.395, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 185}, "mem_gb": 10.14}
|
| 174 |
+
{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16646544051487, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.389, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 185}, "mem_gb": 10.03}
|
| 175 |
+
{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11353683587399621, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.322, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 174}, "mem_gb": 10.07}
|
| 176 |
+
{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11869697581833849, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.298, "comp_len": 701.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.0, "frames": {"chat": 171}, "mem_gb": 10.02}
|
| 177 |
+
{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12181192686005185, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.33, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 176}, "mem_gb": 10.08}
|
| 178 |
+
{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11293597242869437, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.285, "comp_len": 697.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 172}, "mem_gb": 10.14}
|
| 179 |
+
{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11091728310938924, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.381, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 181}, "mem_gb": 10.08}
|
| 180 |
+
[eval step 90] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - We have a plane passing through the origin \\(O(0,0,0)\\) and intersecting the \\(x\\)-axis, \\(y\\)-axis, and \\(z'
|
| 181 |
+
{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12129036977818856, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.441, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 186}, "mem_gb": 10.14}
|
| 182 |
+
{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11895830290373414, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.307, "comp_len": 670.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 179}, "mem_gb": 10.11}
|
| 183 |
+
{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10679319557681059, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.4, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 180}, "mem_gb": 10.02}
|
| 184 |
+
{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10921692143095037, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.376, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 178}, "mem_gb": 10.01}
|
| 185 |
+
{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11740259663785498, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.54, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 200}, "mem_gb": 10.0}
|
| 186 |
+
{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11450160081606979, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.595, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 205}, "mem_gb": 10.0}
|
| 187 |
+
{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15538554082823297, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.291, "comp_len": 685.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 175}, "mem_gb": 10.13}
|
| 188 |
+
{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14038610968940582, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.359, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 181}, "mem_gb": 10.04}
|
| 189 |
+
{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1295151626690291, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.359, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 181}, "mem_gb": 10.03}
|
| 190 |
+
{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10876813413947821, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.316, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.8, "frames": {"chat": 177}, "mem_gb": 10.02}
|
| 191 |
+
[eval step 100] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - The plane passes through the origin \\(O(0,0,0)\\) and intersects the \\(x\\)-axis, \\(y\\)-axis, and \\(z\\)-axis a'
|
| 192 |
+
checkpoint snapshot queued -> outputs/healed/correctness_ab/glean_keep25_nogold_s1224/step0100
|
| 193 |
+
wandb: updating run metadata
|
| 194 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 195 |
+
wandb:
|
| 196 |
+
wandb: Run history:
|
| 197 |
+
wandb: comp_len ▂▃▃▅▆▅▅▃▃▃▇█▄▂▅▃▄▆▅█▂▅▁▅▆▃▆▄▄▇▄▅▅▇▆▆▆▆▂▆
|
| 198 |
+
wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▂▂▂▂▂▃▃▃▄▄▄▄▄▄▅▅▅▅▆▆▆▆▆▇▇▇▇▇████
|
| 199 |
+
wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁█████████████████████
|
| 200 |
+
wandb: finish_rate ▄▅▅▇▆▄▃▅▄▄▇▅▅▅▄▅▃▇▃▁▆▄█▆▅▄█▅▃▅▅▃▃▃▃▃▅▄▇▃
|
| 201 |
+
wandb: forward_topk_kl ██▅▃▂▂▂▂▂▂▂▂▂▁▂▁▁▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 202 |
+
wandb: grad_norm █▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 203 |
+
wandb: lr ▁▃▄▆▇███████████████████████████████████
|
| 204 |
+
wandb: mem_gb ▁▄▁▂▁▂▂▅▁▃▁▂▄▆▅█▂▅▄▃▁▁▆▃▄▅▁▂▁▁▁▅▁▂▂▆▄▆▅▂
|
| 205 |
+
wandb: step ▁▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▃▃▃▄▄▄▄▄▄▄▅▅▅▅▅▆▆▆▆▇▇███
|
| 206 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 207 |
+
wandb: +3 ...
|
| 208 |
+
wandb:
|
| 209 |
+
wandb: Run summary:
|
| 210 |
+
wandb: comp_len 678
|
| 211 |
+
wandb: cumulative_loss_tokens 12000000
|
| 212 |
+
wandb: epoch 1
|
| 213 |
+
wandb: finish_rate 0.316
|
| 214 |
+
wandb: forward_topk_kl 0.10877
|
| 215 |
+
wandb: grad_norm 0.36523
|
| 216 |
+
wandb: lr 3e-05
|
| 217 |
+
wandb: mem_gb 10.02
|
| 218 |
+
wandb: step 100
|
| 219 |
+
wandb: t_data_s 0
|
| 220 |
+
wandb: +4 ...
|
| 221 |
+
wandb:
|
| 222 |
+
wandb: 🚀 View run glean-nogold-keep25-s1224 at: https://wandb.ai/hbfreed/glean-grid/runs/rgbwy6bs
|
| 223 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid
|
| 224 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 225 |
+
wandb: Find logs at: outputs/healed/correctness_ab/glean_keep25_nogold_s1224/wandb/run-20260717_044202-rgbwy6bs/logs
|
| 226 |
+
{
|
| 227 |
+
"correct": 542,
|
| 228 |
+
"accuracy": 0.41091736163760423,
|
| 229 |
+
"finished": 1243,
|
| 230 |
+
"finish_rate": 0.9423805913570887,
|
| 231 |
+
"mean_completion_tokens": 201.3229719484458
|
| 232 |
+
}
|
| 233 |
+
saved item-level results -> outputs/evals/correctness_ab/glean_keep25_nogold_s1224_step100_chat.json
|
healed/correctness_ab/glean_keep25_nogold_s1225.console.log
ADDED
|
@@ -0,0 +1,232 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 6 |
+
wandb: Run data is saved locally in outputs/healed/correctness_ab/glean_keep25_nogold_s1225/wandb/run-20260717_042106-qheyj0qa
|
| 7 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 8 |
+
wandb: Syncing run glean-nogold-keep25-s1225
|
| 9 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 10 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/qheyj0qa
|
| 11 |
+
9918 cached top-128 chat trajectories / 6,476,712 unique tokens | 53 steps/epoch | 150 total steps | student params 2.09B | teacher overlap=False
|
| 12 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2887100736036896, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 87.0, "lr": 6e-06, "finish_rate": 0.266, "comp_len": 693.6, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 37.2, "frames": {"chat": 173}, "mem_gb": 10.06}
|
| 13 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 14 |
+
[eval step 1] sample: '\n@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@'
|
| 15 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2290259738907219, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 86.5, "lr": 9e-06, "finish_rate": 0.558, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 206}, "mem_gb": 10.01}
|
| 16 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.1046923981601993, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 56.0, "lr": 1.2e-05, "finish_rate": 0.523, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 193}, "mem_gb": 10.01}
|
| 17 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.819205822041134, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 13.3125, "lr": 1.5e-05, "finish_rate": 0.312, "comp_len": 693.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 173}, "mem_gb": 10.16}
|
| 18 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7182254951643448, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 12.875, "lr": 1.8e-05, "finish_rate": 0.452, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 188}, "mem_gb": 10.02}
|
| 19 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6411436942835649, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 6.40625, "lr": 2.1e-05, "finish_rate": 0.357, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 182}, "mem_gb": 10.02}
|
| 20 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5608555964551866, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 3.75, "lr": 2.4e-05, "finish_rate": 0.578, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 204}, "mem_gb": 10.01}
|
| 21 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4863458348430693, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 2.203125, "lr": 2.7000000000000002e-05, "finish_rate": 0.473, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 184}, "mem_gb": 10.02}
|
| 22 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4289485489733517, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 1.8125, "lr": 3e-05, "finish_rate": 0.392, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 181}, "mem_gb": 10.13}
|
| 23 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.37692274317281943, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 1.6015625, "lr": 3e-05, "finish_rate": 0.51, "comp_len": 606.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 198}, "mem_gb": 10.06}
|
| 24 |
+
[eval step 10] sample: "To solve this problem, we need to understand the relationship between the teams and the constraints given by the problem.\n\nLet's denote the teams as \\( T_1, T_2, \\ldots, T_n \\).\n\nGiven that for any tw"
|
| 25 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3765609692680339, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 1.2734375, "lr": 3e-05, "finish_rate": 0.505, "comp_len": 625.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 192}, "mem_gb": 10.09}
|
| 26 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3316520971429845, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 0.96875, "lr": 3e-05, "finish_rate": 0.386, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 184}, "mem_gb": 10.01}
|
| 27 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28011969484662014, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.8203125, "lr": 3e-05, "finish_rate": 0.603, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.6, "frames": {"chat": 209}, "mem_gb": 10.01}
|
| 28 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2982797584660351, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.7890625, "lr": 3e-05, "finish_rate": 0.251, "comp_len": 701.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 171}, "mem_gb": 10.12}
|
| 29 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28002098710015416, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.80859375, "lr": 3e-05, "finish_rate": 0.384, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 185}, "mem_gb": 10.02}
|
| 30 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25162537380978467, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.6953125, "lr": 3e-05, "finish_rate": 0.378, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 180}, "mem_gb": 10.01}
|
| 31 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2864169784175853, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 1.46875, "lr": 3e-05, "finish_rate": 0.436, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 188}, "mem_gb": 10.04}
|
| 32 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2514801525355627, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.6875, "lr": 3e-05, "finish_rate": 0.482, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 195}, "mem_gb": 10.06}
|
| 33 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2297573397848755, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.65234375, "lr": 3e-05, "finish_rate": 0.562, "comp_len": 597.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 201}, "mem_gb": 10.01}
|
| 34 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21393080709750453, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.66796875, "lr": 3e-05, "finish_rate": 0.418, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 177}, "mem_gb": 10.02}
|
| 35 |
+
[eval step 20] sample: 'To solve this problem, we need to understand the given condition and the structure of the teams. The condition states that for any two teams, there is always a third team that has defeated both of the'
|
| 36 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.241446342420578, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.64453125, "lr": 3e-05, "finish_rate": 0.444, "comp_len": 641.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 187}, "mem_gb": 10.02}
|
| 37 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22229577187839897, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.6796875, "lr": 3e-05, "finish_rate": 0.421, "comp_len": 631.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 190}, "mem_gb": 10.02}
|
| 38 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21803912110937138, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.59375, "lr": 3e-05, "finish_rate": 0.528, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 197}, "mem_gb": 10.02}
|
| 39 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21678726400832335, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.503, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 193}, "mem_gb": 10.02}
|
| 40 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24970855500604958, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.75390625, "lr": 3e-05, "finish_rate": 0.291, "comp_len": 697.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 172}, "mem_gb": 10.14}
|
| 41 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2177845759611577, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.5859375, "lr": 3e-05, "finish_rate": 0.328, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 174}, "mem_gb": 10.08}
|
| 42 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21297066775839776, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.423, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 182}, "mem_gb": 10.1}
|
| 43 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19239520111481348, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.419, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 186}, "mem_gb": 10.02}
|
| 44 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1812352752689893, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.415, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 188}, "mem_gb": 10.01}
|
| 45 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19375898362770677, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.65625, "lr": 3e-05, "finish_rate": 0.413, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 184}, "mem_gb": 10.03}
|
| 46 |
+
[eval step 30] sample: 'To solve this problem, we need to understand the relationship between the teams and the requirement that for any two teams, there is always a third team that has defeated both of these teams. This pro'
|
| 47 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23041095650084317, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.625, "lr": 3e-05, "finish_rate": 0.395, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 185}, "mem_gb": 10.14}
|
| 48 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 49 |
+
warnings.warn('Grouped GEMM not available.')
|
| 50 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 51 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 52 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 53 |
+
wandb: Run data is saved locally in outputs/healed/correctness_ab/glean_keep25_nogold_s1225/wandb/run-20260717_043837-nt6a78js
|
| 54 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 55 |
+
wandb: Syncing run glean-nogold-keep25-s1225
|
| 56 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 57 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/nt6a78js
|
| 58 |
+
9918 cached top-128 chat trajectories / 6,476,712 unique tokens | 53 steps/epoch | 100 total steps | student params 2.09B | teacher overlap=False
|
| 59 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2887100736036896, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 86.5, "lr": 6e-06, "finish_rate": 0.266, "comp_len": 693.6, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 36.9, "frames": {"chat": 173}, "mem_gb": 10.06}
|
| 60 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 61 |
+
[eval step 1] sample: '\nU1 U1\n\nThe exact number of teams involved in the tournament, however, the exact number of teams involved in the battle, was not exactly between the two teams.\n\nThe exact number of teams involved in t'
|
| 62 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2311958147322137, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 85.5, "lr": 9e-06, "finish_rate": 0.558, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 206}, "mem_gb": 10.01}
|
| 63 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.1080360449726383, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 57.0, "lr": 1.2e-05, "finish_rate": 0.523, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 193}, "mem_gb": 10.01}
|
| 64 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.820142435499529, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 13.625, "lr": 1.5e-05, "finish_rate": 0.312, "comp_len": 693.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 173}, "mem_gb": 10.16}
|
| 65 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7156207797035575, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 10.25, "lr": 1.8e-05, "finish_rate": 0.452, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 188}, "mem_gb": 10.02}
|
| 66 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 67 |
+
warnings.warn('Grouped GEMM not available.')
|
| 68 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 69 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 70 |
+
wandb: setting up run fz6bzzje
|
| 71 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 72 |
+
wandb: Run data is saved locally in outputs/healed/correctness_ab/glean_keep25_nogold_s1225/wandb/run-20260717_044202-fz6bzzje
|
| 73 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 74 |
+
wandb: Syncing run glean-nogold-keep25-s1225
|
| 75 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 76 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/fz6bzzje
|
| 77 |
+
9918 cached top-128 chat trajectories / 6,476,712 unique tokens | 53 steps/epoch | 100 total steps | student params 2.09B | teacher overlap=False
|
| 78 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2887100736036896, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 87.0, "lr": 6e-06, "finish_rate": 0.266, "comp_len": 693.6, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 37.1, "frames": {"chat": 173}, "mem_gb": 10.06}
|
| 79 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 80 |
+
[eval step 1] sample: '\nU=ex=2\n\n|\n\n|\n\n|\n\n|\n\n|\n\n|\n\n|\n\n**\n\n**\n\n**\n\n**\n\n**\n\n **\n\n **\n\n **\n\n **\n\n **\n\n **\n\n **\n\n **\n\n '
|
| 81 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2348659080098072, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 87.5, "lr": 9e-06, "finish_rate": 0.558, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 206}, "mem_gb": 10.01}
|
| 82 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.0968672953034442, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 53.0, "lr": 1.2e-05, "finish_rate": 0.523, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 193}, "mem_gb": 10.01}
|
| 83 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8176534775187572, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 13.3125, "lr": 1.5e-05, "finish_rate": 0.312, "comp_len": 693.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 173}, "mem_gb": 10.16}
|
| 84 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7131726413421333, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 13.1875, "lr": 1.8e-05, "finish_rate": 0.452, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 188}, "mem_gb": 10.02}
|
| 85 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6390248984622459, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 5.46875, "lr": 2.1e-05, "finish_rate": 0.357, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 182}, "mem_gb": 10.02}
|
| 86 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.558639612677445, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 3.46875, "lr": 2.4e-05, "finish_rate": 0.578, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.0, "frames": {"chat": 204}, "mem_gb": 10.01}
|
| 87 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.48640985094234346, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 2.34375, "lr": 2.7000000000000002e-05, "finish_rate": 0.473, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 184}, "mem_gb": 10.02}
|
| 88 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.42916130435218414, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 1.8984375, "lr": 3e-05, "finish_rate": 0.392, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 181}, "mem_gb": 10.13}
|
| 89 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.37568598748271664, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 1.4140625, "lr": 3e-05, "finish_rate": 0.51, "comp_len": 606.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 198}, "mem_gb": 10.06}
|
| 90 |
+
[eval step 10] sample: "To solve this problem, we need to understand the relationship between the teams and the conditions given. Let's denote the teams as \\(T_1, T_2, \\ldots, T_n\\).\n\nGiven:\n1. For any two teams \\(T_i\\) and "
|
| 91 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.37909320481394726, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 1.8125, "lr": 3e-05, "finish_rate": 0.505, "comp_len": 625.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 192}, "mem_gb": 10.09}
|
| 92 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3334877675415327, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 1.03125, "lr": 3e-05, "finish_rate": 0.386, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 184}, "mem_gb": 10.01}
|
| 93 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28235082323973376, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.83203125, "lr": 3e-05, "finish_rate": 0.603, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 209}, "mem_gb": 10.01}
|
| 94 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.30038229855448007, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.79296875, "lr": 3e-05, "finish_rate": 0.251, "comp_len": 701.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 171}, "mem_gb": 10.12}
|
| 95 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28202638405002656, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.8359375, "lr": 3e-05, "finish_rate": 0.384, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 185}, "mem_gb": 10.02}
|
| 96 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25343188057181737, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.7265625, "lr": 3e-05, "finish_rate": 0.378, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 180}, "mem_gb": 10.01}
|
| 97 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.285179545312499, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 1.046875, "lr": 3e-05, "finish_rate": 0.436, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 188}, "mem_gb": 10.04}
|
| 98 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2523546666380018, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.482, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 195}, "mem_gb": 10.06}
|
| 99 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23041357446263233, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.64453125, "lr": 3e-05, "finish_rate": 0.562, "comp_len": 597.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 201}, "mem_gb": 10.01}
|
| 100 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2146183416655908, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.66796875, "lr": 3e-05, "finish_rate": 0.418, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 177}, "mem_gb": 10.02}
|
| 101 |
+
[eval step 20] sample: 'To solve this problem, we need to understand the given condition and use it to determine the minimum number of teams.\n\n**Problem Statement:**\nFor any two teams, there is always a third team that has d'
|
| 102 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24236179417570433, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.65234375, "lr": 3e-05, "finish_rate": 0.444, "comp_len": 641.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 187}, "mem_gb": 10.02}
|
| 103 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22290802211184055, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.421, "comp_len": 631.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 190}, "mem_gb": 10.02}
|
| 104 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21903585352798302, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.58984375, "lr": 3e-05, "finish_rate": 0.528, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 197}, "mem_gb": 10.02}
|
| 105 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21954521602280438, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.503, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 193}, "mem_gb": 10.02}
|
| 106 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2522053220824028, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.62890625, "lr": 3e-05, "finish_rate": 0.291, "comp_len": 697.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 172}, "mem_gb": 10.14}
|
| 107 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22035273789850374, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.58984375, "lr": 3e-05, "finish_rate": 0.328, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 174}, "mem_gb": 10.08}
|
| 108 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21515670088684807, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.423, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 182}, "mem_gb": 10.1}
|
| 109 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1933906206143399, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.419, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 186}, "mem_gb": 10.02}
|
| 110 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18180853872733813, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.52734375, "lr": 3e-05, "finish_rate": 0.415, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 188}, "mem_gb": 10.01}
|
| 111 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19501940278882782, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.413, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 184}, "mem_gb": 10.03}
|
| 112 |
+
[eval step 30] sample: 'To solve this problem, we need to understand the relationship between the teams and the given condition. The condition states that for any two teams, there is always a third team that has defeated bot'
|
| 113 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23210855656502147, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.66796875, "lr": 3e-05, "finish_rate": 0.395, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 185}, "mem_gb": 10.14}
|
| 114 |
+
{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25734003331394245, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.8828125, "lr": 3e-05, "finish_rate": 0.389, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 185}, "mem_gb": 10.03}
|
| 115 |
+
{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19854674719596901, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.322, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 174}, "mem_gb": 10.07}
|
| 116 |
+
{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2014819847221176, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.298, "comp_len": 701.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 171}, "mem_gb": 10.02}
|
| 117 |
+
{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20032484480006, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.52734375, "lr": 3e-05, "finish_rate": 0.33, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 176}, "mem_gb": 10.08}
|
| 118 |
+
{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18328062360429515, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.60546875, "lr": 3e-05, "finish_rate": 0.285, "comp_len": 697.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 172}, "mem_gb": 10.14}
|
| 119 |
+
{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18418232708144933, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.490234375, "lr": 3e-05, "finish_rate": 0.381, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 181}, "mem_gb": 10.08}
|
| 120 |
+
{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1915627242418627, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.441, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 186}, "mem_gb": 10.14}
|
| 121 |
+
{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18639770591290047, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.307, "comp_len": 670.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 179}, "mem_gb": 10.11}
|
| 122 |
+
{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16870365651945274, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.478515625, "lr": 3e-05, "finish_rate": 0.4, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 180}, "mem_gb": 10.02}
|
| 123 |
+
[eval step 40] sample: 'To solve this problem, we need to understand the relationship between the teams and the given condition. The condition states that for any two teams, there is always a third team that has defeated bot'
|
| 124 |
+
{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17494028390347957, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.376, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 178}, "mem_gb": 10.01}
|
| 125 |
+
{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18135482779555023, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.54, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 200}, "mem_gb": 10.0}
|
| 126 |
+
{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1698221223142619, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.595, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.7, "frames": {"chat": 205}, "mem_gb": 10.0}
|
| 127 |
+
{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24732830818984658, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 1.53125, "lr": 3e-05, "finish_rate": 0.291, "comp_len": 685.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.6, "frames": {"chat": 175}, "mem_gb": 10.13}
|
| 128 |
+
{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19134444058078032, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.359, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 181}, "mem_gb": 10.04}
|
| 129 |
+
{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18111671656587472, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.359, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 181}, "mem_gb": 10.03}
|
| 130 |
+
{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16490637171212583, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.316, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 177}, "mem_gb": 10.02}
|
| 131 |
+
{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1748441598246495, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.333, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 174}, "mem_gb": 10.06}
|
| 132 |
+
{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.14833313943600904, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.352, "comp_len": 670.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 179}, "mem_gb": 10.03}
|
| 133 |
+
{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1463591554345563, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.441, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 186}, "mem_gb": 10.03}
|
| 134 |
+
[eval step 50] sample: 'To solve this problem, we need to understand the relationship between the teams and the given condition. The condition states that for any two teams, there is always a third team that has defeated bot'
|
| 135 |
+
checkpoint snapshot queued -> outputs/healed/correctness_ab/glean_keep25_nogold_s1225/step0050
|
| 136 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1898500379689038, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.318, "comp_len": 670.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 179}, "mem_gb": 10.1}
|
| 137 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17548359996893753, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.398, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 181}, "mem_gb": 10.02}
|
| 138 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20467903866395354, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.32, "comp_len": 697.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 172}, "mem_gb": 10.09}
|
| 139 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12816076690567035, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.404, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 183}, "mem_gb": 10.01}
|
| 140 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13108981929148236, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.528, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 197}, "mem_gb": 10.06}
|
| 141 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15983940087252607, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.341, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 176}, "mem_gb": 10.1}
|
| 142 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1251446767653649, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.361, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 180}, "mem_gb": 10.15}
|
| 143 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11424264137775948, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.369, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 176}, "mem_gb": 10.08}
|
| 144 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12102168695932874, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.337, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 178}, "mem_gb": 10.07}
|
| 145 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1208479611520345, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.382, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 186}, "mem_gb": 10.02}
|
| 146 |
+
[eval step 60] sample: 'To solve this problem, we need to understand the relationship between the teams and the given condition. The condition states that for any two teams, there is always a third team that has defeated bot'
|
| 147 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13577270310639095, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.348, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 181}, "mem_gb": 10.11}
|
| 148 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11958012209820251, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.374, "comp_len": 670.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 179}, "mem_gb": 10.02}
|
| 149 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11554981378950178, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.409, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 181}, "mem_gb": 10.07}
|
| 150 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14004465222085516, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.237, "comp_len": 710.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 169}, "mem_gb": 10.06}
|
| 151 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11243003231774394, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.466, "comp_len": 634.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 189}, "mem_gb": 10.03}
|
| 152 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10926986344270408, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.443, "comp_len": 625.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 192}, "mem_gb": 10.01}
|
| 153 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10241294047270591, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.442, "comp_len": 631.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 190}, "mem_gb": 10.01}
|
| 154 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10806114035559197, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.404, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 183}, "mem_gb": 10.04}
|
| 155 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12770395100141566, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.523, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 195}, "mem_gb": 10.02}
|
| 156 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13657056293866288, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.345, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 177}, "mem_gb": 10.11}
|
| 157 |
+
[eval step 70] sample: 'To solve this problem, we need to understand the relationship between the teams and the given condition. The condition states that for any two teams, there is always a third team that has defeated bot'
|
| 158 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17303597443383187, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.331, "comp_len": 685.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 175}, "mem_gb": 10.13}
|
| 159 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13155972927684587, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.301, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.8, "frames": {"chat": 176}, "mem_gb": 10.14}
|
| 160 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.128104401379327, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.44, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 184}, "mem_gb": 10.06}
|
| 161 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10710186393441011, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.413, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 184}, "mem_gb": 10.03}
|
| 162 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1180277938674204, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.317, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.6, "frames": {"chat": 180}, "mem_gb": 10.11}
|
| 163 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1246870937185362, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.457, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.1, "frames": {"chat": 188}, "mem_gb": 10.06}
|
| 164 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11382235308621699, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.414, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 181}, "mem_gb": 10.15}
|
| 165 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11452579427060361, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.459, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 183}, "mem_gb": 10.03}
|
| 166 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11041011377591639, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.503, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.6, "frames": {"chat": 191}, "mem_gb": 10.03}
|
| 167 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1179047228925551, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.393, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 183}, "mem_gb": 10.08}
|
| 168 |
+
[eval step 80] sample: 'To solve this problem, we need to understand the relationship between the teams and the given condition. The condition states that for any two teams, there is always a third team that has defeated bot'
|
| 169 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11089377509492139, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.497, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 193}, "mem_gb": 10.01}
|
| 170 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12570614381569128, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.431, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 188}, "mem_gb": 10.01}
|
| 171 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12479648627247661, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.333, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 177}, "mem_gb": 10.07}
|
| 172 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12964587786092113, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.32, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 178}, "mem_gb": 10.11}
|
| 173 |
+
{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12278947785310447, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.435, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 186}, "mem_gb": 10.02}
|
| 174 |
+
{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1423320757540874, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.356, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 177}, "mem_gb": 10.1}
|
| 175 |
+
{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1245929411392038, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.376, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 181}, "mem_gb": 10.14}
|
| 176 |
+
{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14417291571373742, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.403, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 181}, "mem_gb": 10.03}
|
| 177 |
+
{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1563886462825661, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.241, "comp_len": 705.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 170}, "mem_gb": 10.12}
|
| 178 |
+
{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15911726619092126, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.302, "comp_len": 697.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 172}, "mem_gb": 10.09}
|
| 179 |
+
[eval step 90] sample: 'To solve this problem, we need to understand the relationship between the teams and the given condition. The condition states that for any two teams, there is always a third team that has defeated bot'
|
| 180 |
+
{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12926910570338368, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.377, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 183}, "mem_gb": 10.02}
|
| 181 |
+
{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11234747919421643, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.51, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 200}, "mem_gb": 10.01}
|
| 182 |
+
{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10905423085596412, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.528, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.2, "frames": {"chat": 197}, "mem_gb": 10.01}
|
| 183 |
+
{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1181418479707092, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.492, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.3, "frames": {"chat": 197}, "mem_gb": 10.01}
|
| 184 |
+
{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10888866160496448, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.333, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 177}, "mem_gb": 10.02}
|
| 185 |
+
{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11416195124673346, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.492, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 197}, "mem_gb": 10.05}
|
| 186 |
+
{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13205019473402452, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.374, "comp_len": 641.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.4, "frames": {"chat": 187}, "mem_gb": 10.11}
|
| 187 |
+
{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11549462520747135, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.503, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 195}, "mem_gb": 10.19}
|
| 188 |
+
{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10936471597527464, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.379, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 182}, "mem_gb": 10.08}
|
| 189 |
+
{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11783349358976508, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.414, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 181}, "mem_gb": 10.06}
|
| 190 |
+
[eval step 100] sample: 'To solve this problem, we need to understand the relationship between the teams and the condition given:\n\n1. **Understanding the Condition:**\n For any two teams, there is always a third team that ha'
|
| 191 |
+
checkpoint snapshot queued -> outputs/healed/correctness_ab/glean_keep25_nogold_s1225/step0100
|
| 192 |
+
wandb: updating run metadata
|
| 193 |
+
wandb: uploading output.log
|
| 194 |
+
wandb:
|
| 195 |
+
wandb: Run history:
|
| 196 |
+
wandb: comp_len ██▅▆▁▂▇▂█▄▅▅█▆▄▆▁▇█▆▇▆▇▄▆▃▃▅▂▇▅▃▄▆▆▁▂▂▂▆
|
| 197 |
+
wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▅▆▆▆▇▇▇▇████
|
| 198 |
+
wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁█████████████████████
|
| 199 |
+
wandb: finish_rate ██▆▇▆▅▆█▅▇▅▅▂▃▂▇▂▃▃▃▃▄▄▃▄▅▄▇▂▅▆▅▃▄▁▇▇▆▆▅
|
| 200 |
+
wandb: forward_topk_kl █▇▅▄▄▃▂▂▂▂▂▂▂▂▂▁▁▁▁▂▁▁▁▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 201 |
+
wandb: grad_norm ██▅▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 202 |
+
wandb: lr ▁▅▆█████████████████████████████████████
|
| 203 |
+
wandb: mem_gb ▃▁█▁▁▅▁▆▁▁▅▁▂▇▂▄▆▇▂▃▂▁▅█▁▂▁▂▁▆▃█▂▂▄▄▆▅▁▁
|
| 204 |
+
wandb: step ▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▅▅▆▆▆▇▇▇▇▇▇▇▇██
|
| 205 |
+
wandb: t_data_s █▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 206 |
+
wandb: +3 ...
|
| 207 |
+
wandb:
|
| 208 |
+
wandb: Run summary:
|
| 209 |
+
wandb: comp_len 663
|
| 210 |
+
wandb: cumulative_loss_tokens 12000000
|
| 211 |
+
wandb: epoch 1
|
| 212 |
+
wandb: finish_rate 0.414
|
| 213 |
+
wandb: forward_topk_kl 0.11783
|
| 214 |
+
wandb: grad_norm 0.37109
|
| 215 |
+
wandb: lr 3e-05
|
| 216 |
+
wandb: mem_gb 10.06
|
| 217 |
+
wandb: step 100
|
| 218 |
+
wandb: t_data_s 0
|
| 219 |
+
wandb: +4 ...
|
| 220 |
+
wandb:
|
| 221 |
+
wandb: 🚀 View run glean-nogold-keep25-s1225 at: https://wandb.ai/hbfreed/glean-grid/runs/fz6bzzje
|
| 222 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid
|
| 223 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 224 |
+
wandb: Find logs at: outputs/healed/correctness_ab/glean_keep25_nogold_s1225/wandb/run-20260717_044202-fz6bzzje/logs
|
| 225 |
+
{
|
| 226 |
+
"correct": 543,
|
| 227 |
+
"accuracy": 0.4116755117513268,
|
| 228 |
+
"finished": 1237,
|
| 229 |
+
"finish_rate": 0.9378316906747536,
|
| 230 |
+
"mean_completion_tokens": 194.54738438210765
|
| 231 |
+
}
|
| 232 |
+
saved item-level results -> outputs/evals/correctness_ab/glean_keep25_nogold_s1225_step100_chat.json
|
healed/correctness_ab/glean_keep25_nogold_s1226.console.log
ADDED
|
@@ -0,0 +1,187 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 6 |
+
wandb: Run data is saved locally in outputs/healed/correctness_ab/glean_keep25_nogold_s1226/wandb/run-20260717_043837-zz75g5mz
|
| 7 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 8 |
+
wandb: Syncing run glean-nogold-keep25-s1226
|
| 9 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 10 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/zz75g5mz
|
| 11 |
+
9918 cached top-128 chat trajectories / 6,476,712 unique tokens | 53 steps/epoch | 100 total steps | student params 2.09B | teacher overlap=False
|
| 12 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.4053776189560692, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 97.5, "lr": 6e-06, "finish_rate": 0.404, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 183}, "mem_gb": 9.95}
|
| 13 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 14 |
+
[eval step 1] sample: '\n###\nCalure the value of the volume of the sphere in terms of the given volume.\n\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n'
|
| 15 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2949041513338686, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 88.0, "lr": 9e-06, "finish_rate": 0.528, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 197}, "mem_gb": 10.06}
|
| 16 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.0709974341998498, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 45.75, "lr": 1.2e-05, "finish_rate": 0.341, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 176}, "mem_gb": 10.1}
|
| 17 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8591145280803243, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 13.125, "lr": 1.5e-05, "finish_rate": 0.361, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.5, "frames": {"chat": 180}, "mem_gb": 10.15}
|
| 18 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7396658424491683, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 9.125, "lr": 1.8e-05, "finish_rate": 0.369, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 176}, "mem_gb": 10.08}
|
| 19 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 20 |
+
warnings.warn('Grouped GEMM not available.')
|
| 21 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 22 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 23 |
+
wandb: setting up run 6z4964wk
|
| 24 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 25 |
+
wandb: Run data is saved locally in outputs/healed/correctness_ab/glean_keep25_nogold_s1226/wandb/run-20260717_053543-6z4964wk
|
| 26 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 27 |
+
wandb: Syncing run glean-nogold-keep25-s1226
|
| 28 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 29 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/6z4964wk
|
| 30 |
+
9918 cached top-128 chat trajectories / 6,476,712 unique tokens | 53 steps/epoch | 100 total steps | student params 2.09B | teacher overlap=False
|
| 31 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.4053776189560692, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 97.5, "lr": 6e-06, "finish_rate": 0.404, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.5, "frames": {"chat": 183}, "mem_gb": 9.95}
|
| 32 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 33 |
+
[eval step 1] sample: '\n###\nCalure the value of the volume of the sphere in terms of the given volume.\n\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n###\n'
|
| 34 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2876027997732162, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 88.5, "lr": 9e-06, "finish_rate": 0.528, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 197}, "mem_gb": 10.06}
|
| 35 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.0686750302240253, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 45.25, "lr": 1.2e-05, "finish_rate": 0.341, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 176}, "mem_gb": 10.1}
|
| 36 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8577293424571554, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 13.0, "lr": 1.5e-05, "finish_rate": 0.361, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 180}, "mem_gb": 10.15}
|
| 37 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7386274091054996, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 11.8125, "lr": 1.8e-05, "finish_rate": 0.369, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 176}, "mem_gb": 10.08}
|
| 38 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6128853330232203, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 5.4375, "lr": 2.1e-05, "finish_rate": 0.337, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 178}, "mem_gb": 10.07}
|
| 39 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5478658662686745, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 3.40625, "lr": 2.4e-05, "finish_rate": 0.382, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 186}, "mem_gb": 10.02}
|
| 40 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5200080039662619, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 2.40625, "lr": 2.7000000000000002e-05, "finish_rate": 0.348, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 181}, "mem_gb": 10.11}
|
| 41 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4304271098551651, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 1.703125, "lr": 3e-05, "finish_rate": 0.374, "comp_len": 670.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 179}, "mem_gb": 10.02}
|
| 42 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3849343281839043, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 1.8671875, "lr": 3e-05, "finish_rate": 0.409, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 181}, "mem_gb": 10.07}
|
| 43 |
+
[eval step 10] sample: 'To solve this problem, we need to determine the lateral area of the prism \\(ABCD-A_1B_1C_1D_1\\) given the volume constraint of the circumscribed sphere.\n\n### Step-by-Step Solution:\n\n1. **Understand th'
|
| 44 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.33969252007280787, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 1.140625, "lr": 3e-05, "finish_rate": 0.237, "comp_len": 710.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 169}, "mem_gb": 10.06}
|
| 45 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.31460128467356163, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 0.95703125, "lr": 3e-05, "finish_rate": 0.466, "comp_len": 634.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 189}, "mem_gb": 10.03}
|
| 46 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.29610492084274687, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.89453125, "lr": 3e-05, "finish_rate": 0.443, "comp_len": 625.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 192}, "mem_gb": 10.01}
|
| 47 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2603966006518652, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.75390625, "lr": 3e-05, "finish_rate": 0.442, "comp_len": 631.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.2, "frames": {"chat": 190}, "mem_gb": 10.01}
|
| 48 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2561776851742218, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.7109375, "lr": 3e-05, "finish_rate": 0.404, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 183}, "mem_gb": 10.04}
|
| 49 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24956717135173578, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.523, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 195}, "mem_gb": 10.02}
|
| 50 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2901458036405345, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.80859375, "lr": 3e-05, "finish_rate": 0.345, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 177}, "mem_gb": 10.11}
|
| 51 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3188314758660893, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.9296875, "lr": 3e-05, "finish_rate": 0.331, "comp_len": 685.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.7, "frames": {"chat": 175}, "mem_gb": 10.13}
|
| 52 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2700605358498792, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.6796875, "lr": 3e-05, "finish_rate": 0.301, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 176}, "mem_gb": 10.14}
|
| 53 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23110708448539177, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.44, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 184}, "mem_gb": 10.06}
|
| 54 |
+
[eval step 20] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Volume of the Circumscribed Sprimm:**\n The volume \\( V \\) of the circumscribed sphere of a prism \\( ABCD \\) with base \\( AB'
|
| 55 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21657315355601411, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.413, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 184}, "mem_gb": 10.03}
|
| 56 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23063830836229027, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.66015625, "lr": 3e-05, "finish_rate": 0.317, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 180}, "mem_gb": 10.11}
|
| 57 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23469925464317204, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.62890625, "lr": 3e-05, "finish_rate": 0.457, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 188}, "mem_gb": 10.06}
|
| 58 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.218043468376187, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.414, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 181}, "mem_gb": 10.15}
|
| 59 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2128629672969381, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.459, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 183}, "mem_gb": 10.03}
|
| 60 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20367664861020943, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.503, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 191}, "mem_gb": 10.03}
|
| 61 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21151986253938326, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.393, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 183}, "mem_gb": 10.08}
|
| 62 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19561079414015015, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.54296875, "lr": 3e-05, "finish_rate": 0.497, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 193}, "mem_gb": 10.01}
|
| 63 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22534914367552847, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.8046875, "lr": 3e-05, "finish_rate": 0.431, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 188}, "mem_gb": 10.01}
|
| 64 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2242989562381059, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.333, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 177}, "mem_gb": 10.07}
|
| 65 |
+
[eval step 30] sample: 'To solve this problem, we need to determine the lateral area of the prism \\(ABCD-A_1B_1C_1D_1\\) given the volume of the circumscribed sphere and the side length of the square base \\(ABCD\\).\n\n### Step-'
|
| 66 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2049269159200291, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.5546875, "lr": 3e-05, "finish_rate": 0.32, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 178}, "mem_gb": 10.11}
|
| 67 |
+
{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18366255371204268, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.435, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 186}, "mem_gb": 10.02}
|
| 68 |
+
{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22174582999224465, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.57421875, "lr": 3e-05, "finish_rate": 0.356, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 177}, "mem_gb": 10.1}
|
| 69 |
+
{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20923268946620324, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 2.0625, "lr": 3e-05, "finish_rate": 0.376, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 181}, "mem_gb": 10.14}
|
| 70 |
+
{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22509543277261157, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.403, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 181}, "mem_gb": 10.03}
|
| 71 |
+
{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2002033399677525, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.241, "comp_len": 705.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.1, "frames": {"chat": 170}, "mem_gb": 10.12}
|
| 72 |
+
{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22082350779132295, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.302, "comp_len": 697.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 172}, "mem_gb": 10.09}
|
| 73 |
+
{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19501886300239712, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.377, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 183}, "mem_gb": 10.02}
|
| 74 |
+
{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16628255484917512, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.51, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 200}, "mem_gb": 10.01}
|
| 75 |
+
{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1679727828072384, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.528, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.0, "frames": {"chat": 197}, "mem_gb": 10.01}
|
| 76 |
+
[eval step 40] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - The base \\(ABCD\\) is a square with side length 1.\n - The volume of the circumscribed sphere is given as \\('
|
| 77 |
+
{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18217674622870983, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.625, "lr": 3e-05, "finish_rate": 0.492, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 197}, "mem_gb": 10.01}
|
| 78 |
+
{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16570735449300458, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.333, "comp_len": 678.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.3, "frames": {"chat": 177}, "mem_gb": 10.02}
|
| 79 |
+
{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16060386643080662, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.492, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 197}, "mem_gb": 10.05}
|
| 80 |
+
{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19191757358883818, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.374, "comp_len": 641.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 187}, "mem_gb": 10.11}
|
| 81 |
+
{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16134404958238205, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.503, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 195}, "mem_gb": 10.19}
|
| 82 |
+
{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16970602847101787, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.379, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 182}, "mem_gb": 10.08}
|
| 83 |
+
{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17898543695298333, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.414, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 181}, "mem_gb": 10.06}
|
| 84 |
+
{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15849204997606575, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.566, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 205}, "mem_gb": 10.02}
|
| 85 |
+
{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1822745373973002, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.327, "comp_len": 701.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.9, "frames": {"chat": 171}, "mem_gb": 10.07}
|
| 86 |
+
{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1759520141630744, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.385, "comp_len": 670.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 179}, "mem_gb": 10.02}
|
| 87 |
+
[eval step 50] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - The prism \\(ABCD-A_1B_1C_1D_1\\) is a three-dimensional prism with a square base \\(ABCD\\) and a cylinder base'
|
| 88 |
+
checkpoint snapshot queued -> outputs/healed/correctness_ab/glean_keep25_nogold_s1226/step0050
|
| 89 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19802618774436415, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.314, "comp_len": 685.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 175}, "mem_gb": 10.06}
|
| 90 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15751912521384656, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.477, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 195}, "mem_gb": 10.03}
|
| 91 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1641703026596457, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.632, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 209}, "mem_gb": 10.01}
|
| 92 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1473558990299081, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.422, "comp_len": 641.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.0, "frames": {"chat": 187}, "mem_gb": 10.14}
|
| 93 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15429059411846102, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.399, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 183}, "mem_gb": 10.07}
|
| 94 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1338815799669673, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.361, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 180}, "mem_gb": 10.08}
|
| 95 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15240218772764008, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.262, "comp_len": 714.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 168}, "mem_gb": 10.11}
|
| 96 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12685642822155108, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.446, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 186}, "mem_gb": 10.11}
|
| 97 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1092572331411764, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.46, "comp_len": 641.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 187}, "mem_gb": 10.01}
|
| 98 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11882701478246599, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.51, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 196}, "mem_gb": 10.08}
|
| 99 |
+
[eval step 60] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - The prism \\(ABCD-A_1B_1C_1D_1\\) is a three-dimensional prism with a square base \\(ABCD\\) and a cylinder base'
|
| 100 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13024748928720753, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.453, "comp_len": 625.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.6, "frames": {"chat": 192}, "mem_gb": 10.06}
|
| 101 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1266918835929595, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.439453125, "lr": 3e-05, "finish_rate": 0.463, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 188}, "mem_gb": 10.05}
|
| 102 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13329449380965283, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.313, "comp_len": 670.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 179}, "mem_gb": 10.14}
|
| 103 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16003275414941212, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.275, "comp_len": 701.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.4, "frames": {"chat": 171}, "mem_gb": 10.15}
|
| 104 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13834326796711732, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.469, "comp_len": 618.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 194}, "mem_gb": 10.01}
|
| 105 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10992525289449841, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.439, "comp_len": 641.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 187}, "mem_gb": 10.03}
|
| 106 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1138675967075862, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.508, "comp_len": 621.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 193}, "mem_gb": 10.1}
|
| 107 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13605178271041563, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.474609375, "lr": 3e-05, "finish_rate": 0.409, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 186}, "mem_gb": 10.09}
|
| 108 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11334127667775999, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.412, "comp_len": 659.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.4, "frames": {"chat": 182}, "mem_gb": 10.02}
|
| 109 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10760529593772565, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.479, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 188}, "mem_gb": 10.02}
|
| 110 |
+
[eval step 70] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - The base \\(ABCD\\) is a square with side length 1.\n - The volume of the circumscribed sphere is given as \\('
|
| 111 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13352584876716136, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.451, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 184}, "mem_gb": 10.03}
|
| 112 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13080250038939217, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.376, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.7, "frames": {"chat": 178}, "mem_gb": 10.05}
|
| 113 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11949274915338805, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.42, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 188}, "mem_gb": 10.02}
|
| 114 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14220985092079888, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.351, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 185}, "mem_gb": 10.02}
|
| 115 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10660567930415273, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.465, "comp_len": 641.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 187}, "mem_gb": 10.01}
|
| 116 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13641090428301444, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.383, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.6, "frames": {"chat": 183}, "mem_gb": 10.04}
|
| 117 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11374381143084417, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.395, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 185}, "mem_gb": 10.02}
|
| 118 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11519306119512766, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.515, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.8, "frames": {"chat": 200}, "mem_gb": 10.04}
|
| 119 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12888593478205923, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.559, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.6, "frames": {"chat": 204}, "mem_gb": 10.01}
|
| 120 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11103108659991995, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.547, "comp_len": 597.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 201}, "mem_gb": 10.07}
|
| 121 |
+
[eval step 80] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - The base \\(ABCD\\) is a square with side length 1.\n - The volume of the circumscribed sphere (radius \\(r\\))'
|
| 122 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11562186147825172, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.536, "comp_len": 618.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.4, "frames": {"chat": 194}, "mem_gb": 10.04}
|
| 123 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12900640192367135, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.188, "comp_len": 727.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.4, "frames": {"chat": 165}, "mem_gb": 10.14}
|
| 124 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13480680556539446, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.335, "comp_len": 693.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.2, "frames": {"chat": 173}, "mem_gb": 10.08}
|
| 125 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1083122630596161, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.376, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 181}, "mem_gb": 10.06}
|
| 126 |
+
{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13821670041636874, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.389, "comp_len": 666.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 180}, "mem_gb": 10.13}
|
| 127 |
+
{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11734222290553152, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.385, "comp_len": 670.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.0, "frames": {"chat": 179}, "mem_gb": 10.02}
|
| 128 |
+
{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12341033591010298, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.462, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 186}, "mem_gb": 10.02}
|
| 129 |
+
{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10497758730637531, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.384, "comp_len": 648.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 185}, "mem_gb": 10.07}
|
| 130 |
+
{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12846608681647728, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.57, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.1, "frames": {"chat": 200}, "mem_gb": 10.08}
|
| 131 |
+
{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12073328258600086, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.294, "comp_len": 705.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.2, "frames": {"chat": 170}, "mem_gb": 10.07}
|
| 132 |
+
[eval step 90] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - The base \\(ABCD\\) is a square with side length 1.\n - The volume of the circumscribed sphere is \\(\\frac{4}{'
|
| 133 |
+
{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12294692213255912, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.351, "comp_len": 689.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 174}, "mem_gb": 10.03}
|
| 134 |
+
{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1187617301520581, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.324, "comp_len": 681.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.9, "frames": {"chat": 176}, "mem_gb": 10.09}
|
| 135 |
+
{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10896789911640808, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.536, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.6, "frames": {"chat": 196}, "mem_gb": 10.01}
|
| 136 |
+
{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1058782151671437, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.361, "comp_len": 655.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.5, "frames": {"chat": 183}, "mem_gb": 10.02}
|
| 137 |
+
{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12435936674823364, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.303, "comp_len": 674.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.9, "frames": {"chat": 178}, "mem_gb": 10.01}
|
| 138 |
+
{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11211659687568123, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.403, "comp_len": 645.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.9, "frames": {"chat": 186}, "mem_gb": 10.11}
|
| 139 |
+
{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12068710293335219, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.439453125, "lr": 3e-05, "finish_rate": 0.42, "comp_len": 638.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 29.8, "frames": {"chat": 188}, "mem_gb": 10.01}
|
| 140 |
+
{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13799327422278002, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.408, "comp_len": 652.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 184}, "mem_gb": 10.06}
|
| 141 |
+
{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11780948632347087, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.337, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.3, "frames": {"chat": 181}, "mem_gb": 10.1}
|
| 142 |
+
{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13097247989022484, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.381, "comp_len": 663.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.1, "frames": {"chat": 181}, "mem_gb": 10.14}
|
| 143 |
+
[eval step 100] sample: 'To solve this problem, we need to follow these steps:\n\n1. **Understand the Geometry:**\n - The base \\(ABCD\\) is a square with side length 1.\n - The volume of the circumscribed sphere is \\(\\frac{4}{'
|
| 144 |
+
checkpoint snapshot queued -> outputs/healed/correctness_ab/glean_keep25_nogold_s1226/step0100
|
| 145 |
+
wandb: uploading data; updating run metadata
|
| 146 |
+
wandb: uploading data; uploading wandb-summary.json; uploading config.yaml; uploading output.log
|
| 147 |
+
wandb: uploading data
|
| 148 |
+
wandb: uploading history steps 99-99, summary, console lines 112-114
|
| 149 |
+
wandb:
|
| 150 |
+
wandb: Run history:
|
| 151 |
+
wandb: comp_len ▅▆▄▅▅▃▆▆▅▅▃▅▃▆▅▇▅▃▆▄▆▁▅▇▄▃▇▄▅▄▂▃█▅▅▆▃▅▆▅
|
| 152 |
+
wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇███
|
| 153 |
+
wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁█████████████████
|
| 154 |
+
wandb: finish_rate ▇▄▄▅▄▆▆▆▄▃▆▅▇▅▇▃▆▄▅▃▇▄▅▃▅▇▆▃▇▆▅▇█▁▄█▃▃▄▅
|
| 155 |
+
wandb: forward_topk_kl █▆▄▃▂▃▂▂▂▂▂▂▂▂▂▁▂▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 156 |
+
wandb: grad_norm █▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 157 |
+
wandb: lr ▁▃▆█████████████████████████████████████
|
| 158 |
+
wandb: mem_gb ▄▆▄▄▄▁▃▂▆▇▄▂▆▄▂▁█▂▇▁▂���▄▁▄▂▂▁▃▃▃▂▂▅▂▁▂▁▆█
|
| 159 |
+
wandb: step ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▅▅▅▆▆▆▆▆▆▆▇▇▇▇▇▇▇█████
|
| 160 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 161 |
+
wandb: +3 ...
|
| 162 |
+
wandb:
|
| 163 |
+
wandb: Run summary:
|
| 164 |
+
wandb: comp_len 663
|
| 165 |
+
wandb: cumulative_loss_tokens 12000000
|
| 166 |
+
wandb: epoch 1
|
| 167 |
+
wandb: finish_rate 0.381
|
| 168 |
+
wandb: forward_topk_kl 0.13097
|
| 169 |
+
wandb: grad_norm 0.38477
|
| 170 |
+
wandb: lr 3e-05
|
| 171 |
+
wandb: mem_gb 10.14
|
| 172 |
+
wandb: step 100
|
| 173 |
+
wandb: t_data_s 0
|
| 174 |
+
wandb: +4 ...
|
| 175 |
+
wandb:
|
| 176 |
+
wandb: 🚀 View run glean-nogold-keep25-s1226 at: https://wandb.ai/hbfreed/glean-grid/runs/6z4964wk
|
| 177 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid
|
| 178 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 179 |
+
wandb: Find logs at: outputs/healed/correctness_ab/glean_keep25_nogold_s1226/wandb/run-20260717_053543-6z4964wk/logs
|
| 180 |
+
{
|
| 181 |
+
"correct": 561,
|
| 182 |
+
"accuracy": 0.42532221379833207,
|
| 183 |
+
"finished": 1249,
|
| 184 |
+
"finish_rate": 0.9469294920394238,
|
| 185 |
+
"mean_completion_tokens": 204.07960576194085
|
| 186 |
+
}
|
| 187 |
+
saved item-level results -> outputs/evals/correctness_ab/glean_keep25_nogold_s1226_step100_chat.json
|
healed/grid_math/glean_keep25_s1224.console.log
ADDED
|
@@ -0,0 +1,231 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: setting up run 97sk7igo
|
| 6 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 7 |
+
wandb: Run data is saved locally in outputs/healed/grid_math/glean_keep25_s1224/wandb/run-20260716_040623-97sk7igo
|
| 8 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 9 |
+
wandb: Syncing run glean-math-keep25-s1224
|
| 10 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 11 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/97sk7igo
|
| 12 |
+
12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 2.09B | teacher overlap=False
|
| 13 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.4345772517378133, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 114.5, "lr": 6e-06, "finish_rate": 0.907, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.6, "frames": {"chat": 236}, "mem_gb": 9.77}
|
| 14 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 15 |
+
[eval step 1] sample: '\nThe value of $p$ is the sum of $a$ and $b$. Find the value of $a + b$.\n\n\n### The value of $p$ is the sum of $a$ and $b$. Find the value of $a + b$.\n\n\nLet the value of'
|
| 16 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.401542197600007, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 101.0, "lr": 9e-06, "finish_rate": 0.781, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.7, "frames": {"chat": 215}, "mem_gb": 10.01}
|
| 17 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.1813887482275565, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 63.75, "lr": 1.2e-05, "finish_rate": 0.825, "comp_len": 553.0, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 217}, "mem_gb": 9.88}
|
| 18 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.9318435715690255, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 12.875, "lr": 1.5e-05, "finish_rate": 0.8, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.8, "frames": {"chat": 205}, "mem_gb": 9.94}
|
| 19 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7287501879028976, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 8.4375, "lr": 1.8e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.4, "frames": {"chat": 229}, "mem_gb": 9.92}
|
| 20 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7777328455592195, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 23.625, "lr": 2.1e-05, "finish_rate": 0.812, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 223}, "mem_gb": 9.98}
|
| 21 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5918175142496824, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 4.65625, "lr": 2.4e-05, "finish_rate": 0.708, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 202}, "mem_gb": 10.02}
|
| 22 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5632855169591804, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 3.0, "lr": 2.7000000000000002e-05, "finish_rate": 0.77, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 209}, "mem_gb": 9.99}
|
| 23 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.40953814367316665, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 2.03125, "lr": 3e-05, "finish_rate": 0.885, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 227}, "mem_gb": 9.97}
|
| 24 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.38932390790109833, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 1.5703125, "lr": 3e-05, "finish_rate": 0.848, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.4, "frames": {"chat": 230}, "mem_gb": 10.04}
|
| 25 |
+
[eval step 10] sample: 'To solve the problem, we need to determine the values of \\(a\\), \\(b\\), and \\(p\\) given the equations:\n\n1. \\(a + b = k\\)\n2. \\(k + m = p\\)\n3. \\(p + a = r\\)\n4. \\(b'
|
| 26 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3458594021844367, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 1.2265625, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 231}, "mem_gb": 9.89}
|
| 27 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3208039687448492, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 1.0390625, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.5, "frames": {"chat": 245}, "mem_gb": 9.97}
|
| 28 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2981381397678206, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.921875, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 210}, "mem_gb": 9.98}
|
| 29 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3454915365646283, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.98828125, "lr": 3e-05, "finish_rate": 0.758, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 211}, "mem_gb": 9.97}
|
| 30 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.26953837820465365, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.80859375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 543.0, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 221}, "mem_gb": 10.03}
|
| 31 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2559446799742058, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.8671875, "lr": 3e-05, "finish_rate": 0.912, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 250}, "mem_gb": 9.84}
|
| 32 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2738502737318476, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.8046875, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.8, "frames": {"chat": 229}, "mem_gb": 10.02}
|
| 33 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2143578802034259, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.65625, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.3, "frames": {"chat": 250}, "mem_gb": 9.99}
|
| 34 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24581084873105088, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.73828125, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.2, "frames": {"chat": 231}, "mem_gb": 9.87}
|
| 35 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22843340362496675, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.8125, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.6, "frames": {"chat": 224}, "mem_gb": 9.91}
|
| 36 |
+
[eval step 20] sample: 'To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(p\\), and \\(m\\) given the equations:\n\n1. \\(a + b = k\\)\n2. \\(k + m = p\\)\n3. \\(p + a'
|
| 37 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21071218295594057, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.64453125, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.2, "frames": {"chat": 212}, "mem_gb": 9.95}
|
| 38 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1929007501606519, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.5, "frames": {"chat": 238}, "mem_gb": 9.9}
|
| 39 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2146618340227753, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.68359375, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.9, "frames": {"chat": 257}, "mem_gb": 9.78}
|
| 40 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1882299357444669, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.59375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 227}, "mem_gb": 9.97}
|
| 41 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21730685276426376, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.65234375, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 228}, "mem_gb": 10.0}
|
| 42 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21057547978740185, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.9, "frames": {"chat": 233}, "mem_gb": 9.99}
|
| 43 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1898427692937975, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 233}, "mem_gb": 9.99}
|
| 44 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25671816173580786, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.65625, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 197}, "mem_gb": 10.08}
|
| 45 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22320921461253115, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.8, "frames": {"chat": 239}, "mem_gb": 9.84}
|
| 46 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21965685539674012, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.7578125, "lr": 3e-05, "finish_rate": 0.83, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.2, "frames": {"chat": 224}, "mem_gb": 9.88}
|
| 47 |
+
[eval step 30] sample: 'To solve this problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), and \\(p\\) given the equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &= r \\\\\nb'
|
| 48 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18031089373938738, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 217}, "mem_gb": 10.0}
|
| 49 |
+
{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17237651355092723, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.0, "frames": {"chat": 241}, "mem_gb": 10.0}
|
| 50 |
+
{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17687927930454414, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.546875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 218}, "mem_gb": 9.97}
|
| 51 |
+
{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17552895108697314, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 206}, "mem_gb": 9.98}
|
| 52 |
+
{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18765739145452778, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.0, "frames": {"chat": 232}, "mem_gb": 10.03}
|
| 53 |
+
{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19498228631795694, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.771, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.9, "frames": {"chat": 218}, "mem_gb": 10.04}
|
| 54 |
+
{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19024058150469014, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.474609375, "lr": 3e-05, "finish_rate": 0.779, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.6, "frames": {"chat": 213}, "mem_gb": 10.0}
|
| 55 |
+
{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19190885730143636, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.53125, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.8, "frames": {"chat": 248}, "mem_gb": 9.97}
|
| 56 |
+
{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1669260192029178, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.7, "frames": {"chat": 218}, "mem_gb": 10.04}
|
| 57 |
+
{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15053241867311298, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.851, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.2, "frames": {"chat": 221}, "mem_gb": 9.99}
|
| 58 |
+
[eval step 40] sample: 'To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(p\\), and \\(r\\) given the equations:\n\n\\[\na + b = k\n\\]\n\\[\nk + m = p\n\\]\n\\[\np + a ='
|
| 59 |
+
{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15927721466608347, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.894, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 236}, "mem_gb": 9.92}
|
| 60 |
+
{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16593128025457263, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.2, "frames": {"chat": 246}, "mem_gb": 9.84}
|
| 61 |
+
{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16362612133746346, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.4, "frames": {"chat": 234}, "mem_gb": 10.1}
|
| 62 |
+
{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13735881909814973, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.748, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 202}, "mem_gb": 9.98}
|
| 63 |
+
{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15262007395885885, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.7, "frames": {"chat": 217}, "mem_gb": 9.99}
|
| 64 |
+
{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15682636960850407, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.866, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 224}, "mem_gb": 9.99}
|
| 65 |
+
{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17931298827622086, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.51171875, "lr": 3e-05, "finish_rate": 0.753, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 215}, "mem_gb": 10.01}
|
| 66 |
+
{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1443968278159077, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.6, "frames": {"chat": 259}, "mem_gb": 9.92}
|
| 67 |
+
{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1630018111831819, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.829, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 210}, "mem_gb": 9.99}
|
| 68 |
+
{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1874668618524447, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.3, "frames": {"chat": 213}, "mem_gb": 10.05}
|
| 69 |
+
[eval step 50] sample: 'To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(p\\), and \\(k\\) given the equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &='
|
| 70 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep25_s1224/step0050
|
| 71 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15269769354338447, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.0, "frames": {"chat": 222}, "mem_gb": 9.95}
|
| 72 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13562167353965343, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.431640625, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 235}, "mem_gb": 10.01}
|
| 73 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1533853217214346, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 208}, "mem_gb": 9.96}
|
| 74 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1419351307667171, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.6, "frames": {"chat": 191}, "mem_gb": 10.0}
|
| 75 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11651230888419474, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.9, "frames": {"chat": 219}, "mem_gb": 10.0}
|
| 76 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11036905342560882, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.5, "frames": {"chat": 207}, "mem_gb": 10.0}
|
| 77 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15549946540420254, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.4, "frames": {"chat": 208}, "mem_gb": 9.96}
|
| 78 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10701905750250444, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.8, "frames": {"chat": 219}, "mem_gb": 10.0}
|
| 79 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1119194057648691, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.1, "frames": {"chat": 246}, "mem_gb": 9.87}
|
| 80 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14874430537996813, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.6, "frames": {"chat": 206}, "mem_gb": 10.02}
|
| 81 |
+
[eval step 60] sample: 'To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(p\\), and \\(k\\) given the equations:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\np + a &='
|
| 82 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11569037632470329, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 34.3, "frames": {"chat": 233}, "mem_gb": 10.0}
|
| 83 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12268456889608254, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.1, "frames": {"chat": 229}, "mem_gb": 9.87}
|
| 84 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09885425716893126, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 236}, "mem_gb": 9.9}
|
| 85 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12324718012257169, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.2, "frames": {"chat": 239}, "mem_gb": 9.79}
|
| 86 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10336422332112367, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 241}, "mem_gb": 9.91}
|
| 87 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11680785041383157, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 226}, "mem_gb": 9.87}
|
| 88 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10119038897647212, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.6171875, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.9, "frames": {"chat": 234}, "mem_gb": 10.0}
|
| 89 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10523584497369205, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.9, "frames": {"chat": 257}, "mem_gb": 9.99}
|
| 90 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14938643554880593, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.8, "frames": {"chat": 208}, "mem_gb": 10.05}
|
| 91 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12613096526097506, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.0, "frames": {"chat": 211}, "mem_gb": 10.02}
|
| 92 |
+
[eval step 70] sample: "To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(p\\), and \\(r\\) such that the given equations hold true. Let's break down the problem into manageable steps:\n\n1. **Unders"
|
| 93 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1367481416762496, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 227}, "mem_gb": 10.0}
|
| 94 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12594295182231194, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.8, "frames": {"chat": 211}, "mem_gb": 9.98}
|
| 95 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10491307358372336, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.5, "frames": {"chat": 238}, "mem_gb": 10.0}
|
| 96 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10930286093257989, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.0, "frames": {"chat": 237}, "mem_gb": 10.04}
|
| 97 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12709836459675183, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 208}, "mem_gb": 10.04}
|
| 98 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11188181798982745, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 221}, "mem_gb": 10.12}
|
| 99 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11114264213865002, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.9, "frames": {"chat": 232}, "mem_gb": 9.96}
|
| 100 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10996809244149675, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 208}, "mem_gb": 9.99}
|
| 101 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10375189399402589, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.1, "frames": {"chat": 227}, "mem_gb": 9.91}
|
| 102 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11647727334791173, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.9, "frames": {"chat": 221}, "mem_gb": 9.94}
|
| 103 |
+
[eval step 80] sample: "To solve this problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(p\\), and \\(k\\) such that the given equations hold true. Let's break down the problem step-by-step:\n\n1. **Define Variabl"
|
| 104 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09754915808814889, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.5, "frames": {"chat": 232}, "mem_gb": 10.01}
|
| 105 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10922176310662181, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 219}, "mem_gb": 10.01}
|
| 106 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11440609233131012, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.7, "frames": {"chat": 195}, "mem_gb": 10.1}
|
| 107 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11024705316449206, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.0, "frames": {"chat": 216}, "mem_gb": 10.0}
|
| 108 |
+
{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11196330596281526, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 208}, "mem_gb": 9.89}
|
| 109 |
+
{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10395217997139941, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.8, "frames": {"chat": 235}, "mem_gb": 9.88}
|
| 110 |
+
{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11349097539822882, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 225}, "mem_gb": 9.99}
|
| 111 |
+
{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13914468128886073, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 213}, "mem_gb": 10.08}
|
| 112 |
+
{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09369528885387506, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.3, "frames": {"chat": 257}, "mem_gb": 9.76}
|
| 113 |
+
{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12627653089668603, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 212}, "mem_gb": 10.03}
|
| 114 |
+
[eval step 90] sample: "To solve this problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(p\\), and \\(k\\) such that the given equations hold true. Let's break down the problem step-by-step:\n\n1. **Understand the"
|
| 115 |
+
{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10226257944550986, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 221}, "mem_gb": 10.0}
|
| 116 |
+
{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09869740050801386, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 242}, "mem_gb": 10.0}
|
| 117 |
+
{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09069403918914808, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.9, "frames": {"chat": 220}, "mem_gb": 9.96}
|
| 118 |
+
{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09289198616895204, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.9, "frames": {"chat": 240}, "mem_gb": 9.86}
|
| 119 |
+
{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10285916468321035, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 1.140625, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.8, "frames": {"chat": 206}, "mem_gb": 9.99}
|
| 120 |
+
{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11189375975215808, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 226}, "mem_gb": 10.0}
|
| 121 |
+
{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14419980257482579, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.478515625, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.6, "frames": {"chat": 244}, "mem_gb": 9.79}
|
| 122 |
+
{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1254452561319495, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 224}, "mem_gb": 10.01}
|
| 123 |
+
{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09918995484355837, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.2, "frames": {"chat": 271}, "mem_gb": 9.73}
|
| 124 |
+
{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10652838590508328, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 236}, "mem_gb": 10.01}
|
| 125 |
+
[eval step 100] sample: "To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(p\\), and \\(r\\) such that the given equations hold true. Let's break down the problem step-by-step:\n\n1. **Understand the "
|
| 126 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep25_s1224/step0100
|
| 127 |
+
{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10529813103003738, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 232}, "mem_gb": 9.88}
|
| 128 |
+
{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09428196538443057, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.3, "frames": {"chat": 210}, "mem_gb": 9.94}
|
| 129 |
+
{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09178287912054608, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.0, "frames": {"chat": 217}, "mem_gb": 9.9}
|
| 130 |
+
{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1119527683553286, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 224}, "mem_gb": 10.02}
|
| 131 |
+
{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1298952497580399, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 203}, "mem_gb": 9.87}
|
| 132 |
+
{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10422293208353221, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 239}, "mem_gb": 9.97}
|
| 133 |
+
{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07617318357517942, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.5, "frames": {"chat": 254}, "mem_gb": 9.88}
|
| 134 |
+
{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07282634121378263, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 241}, "mem_gb": 9.98}
|
| 135 |
+
{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.108472445377366, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 213}, "mem_gb": 10.01}
|
| 136 |
+
{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08850192172794293, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 221}, "mem_gb": 10.05}
|
| 137 |
+
[eval step 110] sample: 'To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(r\\), and \\(p\\) such that each letter represents a non-zero digit and satisfies the given equations:\n\n\\[\n\\begin{align*}\na'
|
| 138 |
+
{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09740933078067998, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 30.5, "frames": {"chat": 196}, "mem_gb": 10.01}
|
| 139 |
+
{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07808580766382317, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 270}, "mem_gb": 9.82}
|
| 140 |
+
{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07404585122984524, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.9, "frames": {"chat": 216}, "mem_gb": 9.99}
|
| 141 |
+
{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08850063937480251, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.3, "frames": {"chat": 200}, "mem_gb": 9.96}
|
| 142 |
+
{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07580333673724284, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 206}, "mem_gb": 9.91}
|
| 143 |
+
{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0730737258046555, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.1, "frames": {"chat": 234}, "mem_gb": 9.95}
|
| 144 |
+
{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08553545155295482, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 31.9, "frames": {"chat": 215}, "mem_gb": 9.96}
|
| 145 |
+
{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07238297607673642, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 255}, "mem_gb": 9.94}
|
| 146 |
+
{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07544196644419184, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.0, "frames": {"chat": 250}, "mem_gb": 9.82}
|
| 147 |
+
{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07670339237060397, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.6, "frames": {"chat": 242}, "mem_gb": 9.99}
|
| 148 |
+
[eval step 120] sample: 'To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(r\\), and \\(p\\) such that each letter represents a non-zero digit and satisfies the given equations:\n\n\\[\n\\begin{align*}\na'
|
| 149 |
+
{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0937881386723059, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.5, "frames": {"chat": 199}, "mem_gb": 10.0}
|
| 150 |
+
{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11198460068336377, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.5, "frames": {"chat": 208}, "mem_gb": 10.04}
|
| 151 |
+
{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07519433575168562, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.8, "frames": {"chat": 208}, "mem_gb": 9.97}
|
| 152 |
+
{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09269439431062589, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.5, "frames": {"chat": 209}, "mem_gb": 10.12}
|
| 153 |
+
{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06901245591950914, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 235}, "mem_gb": 9.96}
|
| 154 |
+
{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07502137167186787, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.1, "frames": {"chat": 204}, "mem_gb": 9.95}
|
| 155 |
+
{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09983784483978525, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.7, "frames": {"chat": 208}, "mem_gb": 10.01}
|
| 156 |
+
{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07167404105997023, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.4, "frames": {"chat": 240}, "mem_gb": 10.0}
|
| 157 |
+
{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07390924430101489, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.3046875, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 246}, "mem_gb": 9.99}
|
| 158 |
+
{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08214766127246742, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 243}, "mem_gb": 9.81}
|
| 159 |
+
[eval step 130] sample: 'To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(r\\), and \\(p\\) such that each letter represents a non-zero digit and satisfies the given equations:\n\n\\[\n\\begin{align*}\na'
|
| 160 |
+
{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09511748868422583, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 32.2, "frames": {"chat": 208}, "mem_gb": 10.01}
|
| 161 |
+
{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09182360006729141, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.4, "frames": {"chat": 219}, "mem_gb": 10.0}
|
| 162 |
+
{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10137045447360724, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 211}, "mem_gb": 10.01}
|
| 163 |
+
{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08328102445462719, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.4, "frames": {"chat": 232}, "mem_gb": 9.97}
|
| 164 |
+
{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10078765981163208, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.4, "frames": {"chat": 214}, "mem_gb": 10.01}
|
| 165 |
+
{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07711706876040747, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.1, "frames": {"chat": 226}, "mem_gb": 9.9}
|
| 166 |
+
{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07596234880803773, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.3046875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.6, "frames": {"chat": 210}, "mem_gb": 10.01}
|
| 167 |
+
{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06859002932261986, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.6, "frames": {"chat": 218}, "mem_gb": 9.83}
|
| 168 |
+
{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06959025802219597, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.3, "frames": {"chat": 233}, "mem_gb": 9.99}
|
| 169 |
+
{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09598405006804193, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.3, "frames": {"chat": 215}, "mem_gb": 10.01}
|
| 170 |
+
[eval step 140] sample: "To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(m\\), \\(r\\), and \\(p\\) such that the given equations hold true. Let's break down the problem step-by-step and use Python with Sy"
|
| 171 |
+
{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08247211161882927, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.0, "frames": {"chat": 233}, "mem_gb": 9.99}
|
| 172 |
+
{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07299311864568542, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.7, "frames": {"chat": 209}, "mem_gb": 9.94}
|
| 173 |
+
{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06838633713191375, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.7, "frames": {"chat": 262}, "mem_gb": 9.87}
|
| 174 |
+
{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07117203206044312, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.9, "frames": {"chat": 249}, "mem_gb": 9.96}
|
| 175 |
+
{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09463274955069646, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.88671875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.3, "frames": {"chat": 227}, "mem_gb": 10.0}
|
| 176 |
+
{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07090324180225531, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.3, "frames": {"chat": 221}, "mem_gb": 9.99}
|
| 177 |
+
{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07361622673333623, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 234}, "mem_gb": 10.01}
|
| 178 |
+
{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06774209337647383, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.8, "frames": {"chat": 213}, "mem_gb": 9.96}
|
| 179 |
+
{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06980254276961399, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.0, "frames": {"chat": 213}, "mem_gb": 9.89}
|
| 180 |
+
{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07166271255910396, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.5, "frames": {"chat": 234}, "mem_gb": 9.92}
|
| 181 |
+
[eval step 150] sample: 'To solve the given system of equations involving the digits \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\), we need to ensure that each digit is a non-zero digit (i.e., \\(1 \\leq a, b, k, m, p \\leq'
|
| 182 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep25_s1224/step0150
|
| 183 |
+
wandb: updating run metadata
|
| 184 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 185 |
+
wandb:
|
| 186 |
+
wandb: Run history:
|
| 187 |
+
wandb: comp_len ▇▃▃▆▄▃▁▃▇▃▆▅▃▃▅█▂▃▁▅▆▄▃▇▂▃▆▅▃▅▁▂▇▆▆▂▅▅▆▃
|
| 188 |
+
wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▃▃▃▃▃▃▃▄▄▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▇▇█████
|
| 189 |
+
wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅████████████
|
| 190 |
+
wandb: finish_rate ▄▁▃▆▇▅█▄▅▅▇▂▆▃▅▇▅▄▃▂▆█▃▅▁▅▅█▄▅▆▂▆▂▂▂▃▅▇▅
|
| 191 |
+
wandb: forward_topk_kl █▆▅▂▂▂▂▂▂▂▂▂▁▂▂▂▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 192 |
+
wandb: grad_norm █▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 193 |
+
wandb: lr ▁███████████████████████████████████████
|
| 194 |
+
wandb: mem_gb ▆▅▅▅▃▅▇▂▅▆▅▅▁█▅▅▃▆▅▅▅█▂▅▅▂▃▂▅▂▁▁█▄▅▅▅▁▄▃
|
| 195 |
+
wandb: step ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▃▃▃▄▄▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇████
|
| 196 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 197 |
+
wandb: +3 ...
|
| 198 |
+
wandb:
|
| 199 |
+
wandb: Run summary:
|
| 200 |
+
wandb: comp_len 512.8
|
| 201 |
+
wandb: cumulative_loss_tokens 18000000
|
| 202 |
+
wandb: epoch 2
|
| 203 |
+
wandb: finish_rate 0.906
|
| 204 |
+
wandb: forward_topk_kl 0.07166
|
| 205 |
+
wandb: grad_norm 0.29688
|
| 206 |
+
wandb: lr 3e-05
|
| 207 |
+
wandb: mem_gb 9.92
|
| 208 |
+
wandb: step 150
|
| 209 |
+
wandb: t_data_s 0
|
| 210 |
+
wandb: +4 ...
|
| 211 |
+
wandb:
|
| 212 |
+
wandb: 🚀 View run glean-math-keep25-s1224 at: https://wandb.ai/hbfreed/glean-grid/runs/97sk7igo
|
| 213 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid
|
| 214 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 215 |
+
wandb: Find logs at: outputs/healed/grid_math/glean_keep25_s1224/wandb/run-20260716_040623-97sk7igo/logs
|
| 216 |
+
{
|
| 217 |
+
"correct": 545,
|
| 218 |
+
"accuracy": 0.4131918119787718,
|
| 219 |
+
"finished": 1283,
|
| 220 |
+
"finish_rate": 0.9727065959059894,
|
| 221 |
+
"mean_completion_tokens": 202.013646702047
|
| 222 |
+
}
|
| 223 |
+
saved item-level results -> outputs/evals/grid_math/glean_keep25_s1224_step100_chat.json
|
| 224 |
+
{
|
| 225 |
+
"correct": 569,
|
| 226 |
+
"accuracy": 0.4313874147081122,
|
| 227 |
+
"finished": 1277,
|
| 228 |
+
"finish_rate": 0.9681576952236542,
|
| 229 |
+
"mean_completion_tokens": 176.21683093252463
|
| 230 |
+
}
|
| 231 |
+
saved item-level results -> outputs/evals/grid_math/glean_keep25_s1224_step150_chat.json
|
healed/grid_math/glean_keep75_s1224.console.log
ADDED
|
@@ -0,0 +1,232 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: setting up run d06of2kq
|
| 6 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 7 |
+
wandb: Run data is saved locally in outputs/healed/grid_math/glean_keep75_s1224/wandb/run-20260716_151441-d06of2kq
|
| 8 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 9 |
+
wandb: Syncing run glean-math-keep75-s1224
|
| 10 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 11 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/d06of2kq
|
| 12 |
+
|
| 13 |
+
12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 5.31B | teacher overlap=False
|
| 14 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03449597280910239, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 0.68359375, "lr": 6e-06, "finish_rate": 0.907, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.1, "frames": {"chat": 236}, "mem_gb": 21.78}
|
| 15 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 16 |
+
[eval step 1] sample: 'To solve the given system of equations, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(r\\) such that each letter represents a non-zero digit and the equations are satisfied.\n\nThe equa'
|
| 17 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06011146203293465, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 0.89453125, "lr": 9e-06, "finish_rate": 0.781, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 215}, "mem_gb": 22.1}
|
| 18 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.050602795191947374, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 0.8671875, "lr": 1.2e-05, "finish_rate": 0.825, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 217}, "mem_gb": 21.98}
|
| 19 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0385761997969045, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 0.69140625, "lr": 1.5e-05, "finish_rate": 0.8, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.4, "frames": {"chat": 205}, "mem_gb": 22.03}
|
| 20 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.028516700802991787, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 0.53515625, "lr": 1.8e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 229}, "mem_gb": 22.01}
|
| 21 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.09759791274744396, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 3.328125, "lr": 2.1e-05, "finish_rate": 0.812, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 223}, "mem_gb": 22.08}
|
| 22 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.037956103402220956, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 0.53125, "lr": 2.4e-05, "finish_rate": 0.708, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 202}, "mem_gb": 22.12}
|
| 23 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04556338287966016, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 0.51953125, "lr": 2.7000000000000002e-05, "finish_rate": 0.77, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 209}, "mem_gb": 22.09}
|
| 24 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03176342047526656, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.885, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 227}, "mem_gb": 22.06}
|
| 25 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03453842100337303, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.848, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 230}, "mem_gb": 22.14}
|
| 26 |
+
[eval step 10] sample: "To solve the given system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that each letter represents a non-zero digit. Let's break down the problem step-by"
|
| 27 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.030750091298886884, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 231}, "mem_gb": 21.99}
|
| 28 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.033893703076041615, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 245}, "mem_gb": 22.06}
|
| 29 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03699809879013337, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 210}, "mem_gb": 22.07}
|
| 30 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03643371430290863, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.758, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 211}, "mem_gb": 22.07}
|
| 31 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03376166817937822, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 221}, "mem_gb": 22.12}
|
| 32 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.025529393225275757, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.912, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 250}, "mem_gb": 21.94}
|
| 33 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.028599764375450712, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 229}, "mem_gb": 22.11}
|
| 34 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.021392232792646005, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 250}, "mem_gb": 22.08}
|
| 35 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.028017116936428162, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 231}, "mem_gb": 21.96}
|
| 36 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.024294521242644015, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 224}, "mem_gb": 22.0}
|
| 37 |
+
[eval step 20] sample: "To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that each letter represents a non-zero digit and the given equations are satisfied.\n\nLet's break dow"
|
| 38 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03302676243637689, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 212}, "mem_gb": 22.04}
|
| 39 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.029354514089599253, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 238}, "mem_gb": 22.0}
|
| 40 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.023834507278733267, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 257}, "mem_gb": 21.88}
|
| 41 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.024567021203033317, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 227}, "mem_gb": 22.07}
|
| 42 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.025116141962097025, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 228}, "mem_gb": 22.09}
|
| 43 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02764747820661093, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 233}, "mem_gb": 22.09}
|
| 44 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.024866033803291308, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 233}, "mem_gb": 22.08}
|
| 45 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.033120558351837096, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 609.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 197}, "mem_gb": 22.18}
|
| 46 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.031153174231007386, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 239}, "mem_gb": 21.93}
|
| 47 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02932017131882409, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.83, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 224}, "mem_gb": 21.98}
|
| 48 |
+
[eval step 30] sample: "To solve the given system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that each letter represents a non-zero digit. Let's break down the problem step-by"
|
| 49 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.028374534969303446, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 217}, "mem_gb": 22.09}
|
| 50 |
+
{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02549648174579876, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 241}, "mem_gb": 22.09}
|
| 51 |
+
{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.021897728413917746, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 218}, "mem_gb": 22.06}
|
| 52 |
+
{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.026583122743371254, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 206}, "mem_gb": 22.07}
|
| 53 |
+
{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.024209661411851022, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 232}, "mem_gb": 22.12}
|
| 54 |
+
{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.030011083026354513, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.771, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 218}, "mem_gb": 22.14}
|
| 55 |
+
{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02819100455886995, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.779, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 213}, "mem_gb": 22.1}
|
| 56 |
+
{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.023883562099013943, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 248}, "mem_gb": 22.07}
|
| 57 |
+
{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.01988168048077884, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 218}, "mem_gb": 22.13}
|
| 58 |
+
{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.022642062246473506, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.851, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 221}, "mem_gb": 22.08}
|
| 59 |
+
[eval step 40] sample: 'To solve the given system of equations, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that each letter represents a non-zero digit and the equations are satisfied.\n\nLet'
|
| 60 |
+
{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.019084759334203165, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.894, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 236}, "mem_gb": 22.02}
|
| 61 |
+
{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02037259669423414, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 246}, "mem_gb": 21.94}
|
| 62 |
+
{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.023210749543830754, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 234}, "mem_gb": 22.19}
|
| 63 |
+
{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.023896971165249123, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.748, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 202}, "mem_gb": 22.07}
|
| 64 |
+
{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.022854564331177, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.7, "frames": {"chat": 217}, "mem_gb": 22.09}
|
| 65 |
+
{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.019078410341955412, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.866, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 224}, "mem_gb": 22.09}
|
| 66 |
+
{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02562954197395593, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.753, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 215}, "mem_gb": 22.1}
|
| 67 |
+
{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.018489539842109663, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 259}, "mem_gb": 22.02}
|
| 68 |
+
{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02005737902369971, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.829, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 210}, "mem_gb": 22.09}
|
| 69 |
+
{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02531634767386131, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 213}, "mem_gb": 22.14}
|
| 70 |
+
[eval step 50] sample: "To solve the problem, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(r\\) such that each letter represents a non-zero digit and the given equations are satisfied.\n\nLet's break down the"
|
| 71 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep75_s1224/step0050
|
| 72 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.020402686214788506, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.5, "frames": {"chat": 222}, "mem_gb": 22.05}
|
| 73 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.021501830867379126, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 235}, "mem_gb": 22.1}
|
| 74 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.021905474028519046, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 208}, "mem_gb": 22.06}
|
| 75 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.014909254243193814, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 191}, "mem_gb": 22.1}
|
| 76 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.014094466440717225, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.2060546875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 219}, "mem_gb": 22.09}
|
| 77 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01839226848000738, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 207}, "mem_gb": 22.1}
|
| 78 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.024518809633422643, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 208}, "mem_gb": 22.05}
|
| 79 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01596627751175935, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 219}, "mem_gb": 22.09}
|
| 80 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0172666053055902, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.2412109375, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 246}, "mem_gb": 21.96}
|
| 81 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.024067298068331244, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 206}, "mem_gb": 22.12}
|
| 82 |
+
[eval step 60] sample: "To solve the problem, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), \\(p\\), and \\(r\\) such that each letter represents a non-zero digit and the given equations are satisfied.\n\nLet's br"
|
| 83 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02249295396681021, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 233}, "mem_gb": 22.1}
|
| 84 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01704160064985044, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.2119140625, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 229}, "mem_gb": 21.96}
|
| 85 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015235947330253354, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.2216796875, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 236}, "mem_gb": 21.99}
|
| 86 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01926329362227116, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 239}, "mem_gb": 21.88}
|
| 87 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.014964974110476518, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 241}, "mem_gb": 22.0}
|
| 88 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016025373026115508, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.2080078125, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 226}, "mem_gb": 21.97}
|
| 89 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.013881851029904404, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.203125, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.2, "frames": {"chat": 234}, "mem_gb": 22.09}
|
| 90 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01412244028544907, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.1884765625, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 257}, "mem_gb": 22.08}
|
| 91 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.022030819237659066, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 208}, "mem_gb": 22.14}
|
| 92 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019699824277381414, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.240234375, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 211}, "mem_gb": 22.11}
|
| 93 |
+
[eval step 70] sample: 'To solve the problem, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that the given equations are satisfied:\n\n\\[\n\\begin{align*}\na + b &= k \\\\\nk + m &= p \\\\\n'
|
| 94 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01967962834225036, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 227}, "mem_gb": 22.1}
|
| 95 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01805762037136592, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 211}, "mem_gb": 22.07}
|
| 96 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015984567630795452, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.2255859375, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 238}, "mem_gb": 22.09}
|
| 97 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.014909437761079366, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.2080078125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 237}, "mem_gb": 22.13}
|
| 98 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019546328436707457, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.244140625, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 208}, "mem_gb": 22.13}
|
| 99 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.014784362400711204, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.2080078125, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 221}, "mem_gb": 22.22}
|
| 100 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01612296600251575, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.2275390625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 232}, "mem_gb": 22.05}
|
| 101 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017424311850770997, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.2421875, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 208}, "mem_gb": 22.09}
|
| 102 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01336280028744368, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.2080078125, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 227}, "mem_gb": 22.01}
|
| 103 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015608607143598298, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 221}, "mem_gb": 22.03}
|
| 104 |
+
[eval step 80] sample: 'To solve the given system of equations for \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\), we will follow these steps:\n\n1. **Understand the Equations:**\n \\[\n \\begin{align*}\n a + b &= k \\\\\n'
|
| 105 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.014382271881537357, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.2080078125, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 232}, "mem_gb": 22.1}
|
| 106 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017395038500472824, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.2333984375, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 219}, "mem_gb": 22.1}
|
| 107 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01632010119668363, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 195}, "mem_gb": 22.19}
|
| 108 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016265608747016328, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.2216796875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 216}, "mem_gb": 22.1}
|
| 109 |
+
{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017849171374470462, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.2421875, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 208}, "mem_gb": 21.98}
|
| 110 |
+
{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.013051833534993541, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.2060546875, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 235}, "mem_gb": 21.97}
|
| 111 |
+
{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015104487207427155, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 225}, "mem_gb": 22.08}
|
| 112 |
+
{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019873610892542637, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.2412109375, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 213}, "mem_gb": 22.18}
|
| 113 |
+
{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.014629409632699875, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 257}, "mem_gb": 21.85}
|
| 114 |
+
{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02160333691588603, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.236328125, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 212}, "mem_gb": 22.12}
|
| 115 |
+
[eval step 90] sample: "To solve this problem, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that the given equations are satisfied. Let's break down the problem step-by-step:\n\n1. **Define Variable"
|
| 116 |
+
{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01650589420837738, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.234375, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 221}, "mem_gb": 22.09}
|
| 117 |
+
{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.013465291119840307, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.1826171875, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 242}, "mem_gb": 22.09}
|
| 118 |
+
{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01404507189298941, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 220}, "mem_gb": 22.05}
|
| 119 |
+
{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01276353889235373, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.1923828125, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 240}, "mem_gb": 21.95}
|
| 120 |
+
{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01450902842770641, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.2109375, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 206}, "mem_gb": 22.08}
|
| 121 |
+
{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01776092460920336, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.21875, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 226}, "mem_gb": 22.09}
|
| 122 |
+
{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02220483287217406, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.24609375, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 244}, "mem_gb": 21.88}
|
| 123 |
+
{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017873613287787884, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.234375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 224}, "mem_gb": 22.1}
|
| 124 |
+
{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.014570516015263274, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.7, "frames": {"chat": 271}, "mem_gb": 21.82}
|
| 125 |
+
{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015648798539291603, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.23046875, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 236}, "mem_gb": 22.1}
|
| 126 |
+
[eval step 100] sample: "To solve the given system of equations, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that each letter represents a non-zero digit and the equations hold true.\n\nLet's break "
|
| 127 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep75_s1224/step0100
|
| 128 |
+
{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01392048208489238, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 232}, "mem_gb": 21.97}
|
| 129 |
+
{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01582302356507086, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.234375, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.7, "frames": {"chat": 210}, "mem_gb": 22.03}
|
| 130 |
+
{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015148047948399714, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.2333984375, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 217}, "mem_gb": 22.0}
|
| 131 |
+
{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016510671148794547, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 224}, "mem_gb": 22.12}
|
| 132 |
+
{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.020347583135644283, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 203}, "mem_gb": 21.96}
|
| 133 |
+
{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.014199156087749482, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 239}, "mem_gb": 22.06}
|
| 134 |
+
{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.008977783386060036, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.1494140625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.7, "frames": {"chat": 254}, "mem_gb": 21.98}
|
| 135 |
+
{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010862475730893979, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.19140625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 241}, "mem_gb": 22.07}
|
| 136 |
+
{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014401656335553464, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.2099609375, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 213}, "mem_gb": 22.1}
|
| 137 |
+
{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014489341427544909, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 221}, "mem_gb": 22.15}
|
| 138 |
+
[eval step 110] sample: "To solve this problem, we need to determine the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that the given equations are satisfied. Let's break down the problem step-by-step:\n\n1. **Understand"
|
| 139 |
+
{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013501541669142898, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.1865234375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.4, "frames": {"chat": 196}, "mem_gb": 22.11}
|
| 140 |
+
{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.008731955103940951, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.13671875, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 270}, "mem_gb": 21.91}
|
| 141 |
+
{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010822844643716235, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.181640625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 216}, "mem_gb": 22.09}
|
| 142 |
+
{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011418462452020807, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.1728515625, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.0, "frames": {"chat": 200}, "mem_gb": 22.06}
|
| 143 |
+
{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010639629699802026, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.1640625, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 206}, "mem_gb": 22.01}
|
| 144 |
+
{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009322874530012874, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.140625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 234}, "mem_gb": 22.04}
|
| 145 |
+
{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01033369288703737, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.1552734375, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 215}, "mem_gb": 22.05}
|
| 146 |
+
{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01243805469731257, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.197265625, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 255}, "mem_gb": 22.03}
|
| 147 |
+
{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010969508257628574, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 250}, "mem_gb": 21.92}
|
| 148 |
+
{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009528974813098709, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.146484375, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 242}, "mem_gb": 22.09}
|
| 149 |
+
[eval step 120] sample: "To solve this problem, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that each letter represents a non-zero digit and the given equations are satisfied.\n\nLet's break down th"
|
| 150 |
+
{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012388043802530349, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.1689453125, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 199}, "mem_gb": 22.1}
|
| 151 |
+
{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015805441350703282, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.2158203125, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 208}, "mem_gb": 22.13}
|
| 152 |
+
{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012433360835929245, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.2177734375, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 208}, "mem_gb": 22.07}
|
| 153 |
+
{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014551944470048572, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 209}, "mem_gb": 22.22}
|
| 154 |
+
{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009675193485268392, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.16796875, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 235}, "mem_gb": 22.05}
|
| 155 |
+
{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011770741490732569, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.1943359375, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 204}, "mem_gb": 22.04}
|
| 156 |
+
{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014998916014590457, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 208}, "mem_gb": 22.1}
|
| 157 |
+
{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011279444927962807, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 240}, "mem_gb": 22.09}
|
| 158 |
+
{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010361070156555312, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 246}, "mem_gb": 22.08}
|
| 159 |
+
{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009969970819191076, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.1513671875, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 243}, "mem_gb": 21.91}
|
| 160 |
+
[eval step 130] sample: "To solve this problem, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(r\\) such that each letter represents a non-zero digit and the given equations are satisfied.\n\nLet's break down th"
|
| 161 |
+
{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01188105622678607, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.162109375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 208}, "mem_gb": 22.1}
|
| 162 |
+
{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013685301415641636, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.1875, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 219}, "mem_gb": 22.1}
|
| 163 |
+
{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014316876399792575, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 211}, "mem_gb": 22.11}
|
| 164 |
+
{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014087832710818232, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.2236328125, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 232}, "mem_gb": 22.07}
|
| 165 |
+
{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01439627331920977, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.1748046875, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 214}, "mem_gb": 22.1}
|
| 166 |
+
{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010959521841653623, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.1640625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 226}, "mem_gb": 21.99}
|
| 167 |
+
{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010081585236514609, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.1787109375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 210}, "mem_gb": 22.1}
|
| 168 |
+
{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010659327802799332, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.1826171875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.4, "frames": {"chat": 218}, "mem_gb": 21.93}
|
| 169 |
+
{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010789314434929595, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.158203125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 233}, "mem_gb": 22.08}
|
| 170 |
+
{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014221973552062022, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 215}, "mem_gb": 22.1}
|
| 171 |
+
[eval step 140] sample: "To solve the given system of equations, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that each letter represents a non-zero digit and the equations hold true.\n\nLet's break "
|
| 172 |
+
{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010910056351528813, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.169921875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 233}, "mem_gb": 22.08}
|
| 173 |
+
{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011350942074880004, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.1953125, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 209}, "mem_gb": 22.03}
|
| 174 |
+
{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.00926427476175207, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.158203125, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 262}, "mem_gb": 21.97}
|
| 175 |
+
{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011766176097157102, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.1787109375, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 249}, "mem_gb": 22.06}
|
| 176 |
+
{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013218875605349118, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.1904296875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 227}, "mem_gb": 22.09}
|
| 177 |
+
{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009844068101909943, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.150390625, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 221}, "mem_gb": 22.09}
|
| 178 |
+
{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010795413363662859, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.1689453125, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 234}, "mem_gb": 22.1}
|
| 179 |
+
{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01047869203084847, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.197265625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 213}, "mem_gb": 22.05}
|
| 180 |
+
{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009803685631471066, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.1435546875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 213}, "mem_gb": 21.99}
|
| 181 |
+
{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.00903664315190787, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.1455078125, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 234}, "mem_gb": 22.02}
|
| 182 |
+
[eval step 150] sample: "To solve the given system of equations, we need to find the values of \\(a\\), \\(b\\), \\(k\\), \\(m\\), and \\(p\\) such that each letter represents a non-zero digit and the equations are satisfied.\n\nLet's br"
|
| 183 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep75_s1224/step0150
|
| 184 |
+
wandb: updating run metadata
|
| 185 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 186 |
+
wandb:
|
| 187 |
+
wandb: Run history:
|
| 188 |
+
wandb: comp_len ▄▄▆▄▂▃▆▅▅▇▁▆▆▅▆▆▃▄▃▃▃▆█▅▅▆▄▄▆▅▅▇▅▃▆▅▅▃▆▅
|
| 189 |
+
wandb: cumulative_loss_tokens ▁▁▂▂▂▂▂▂▂▂▃▃▃▃▃▃▃▃▄▄▄▄▄▅▅▆▆▆▆▆▇▇▇▇▇▇████
|
| 190 |
+
wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅████████
|
| 191 |
+
wandb: finish_rate ▃▅▃▄▆▄▂▅▃▇▃▂▄█▆█▃▄▅▃▁█▆▅▇▄▅▅▆▂▂▂▂▃▆▅▃▇▅▆
|
| 192 |
+
wandb: forward_topk_kl ▅▄▃▃█▃▃▃▃▂▂▂▂▂▃▂▂▂▁▂▁▁▁▂▂▁▁▂▁▁▂▁▁▁▁▁▁▁▁▁
|
| 193 |
+
wandb: grad_norm █▅▃▃▃▂▃▃▃▃▃▂▂▂▂▂▂▃▂▂▂▂▂▂▂▂▂▁▁▁▁▂▁▂▂▂▁▁▁▁
|
| 194 |
+
wandb: lr ▁███████████████████████████████████████
|
| 195 |
+
wandb: mem_gb ▄▆▆▇▅▆▂▆▁▆▆▆▇▅▇▄▇▅▆▆▃▆▆▇▅▆██▆▆▆▅▅▅▂▂▂▆▄▄
|
| 196 |
+
wandb: step ▁▁▁▁▁▂▂▂▃▃▃▃▄▄▄▄▄▄▄▅▅▅▅▅▅▅▅▅▆▆▆▆▆▆▇█████
|
| 197 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 198 |
+
wandb: +3 ...
|
| 199 |
+
wandb:
|
| 200 |
+
wandb: Run summary:
|
| 201 |
+
wandb: comp_len 512.8
|
| 202 |
+
wandb: cumulative_loss_tokens 18000000
|
| 203 |
+
wandb: epoch 2
|
| 204 |
+
wandb: finish_rate 0.906
|
| 205 |
+
wandb: forward_topk_kl 0.00904
|
| 206 |
+
wandb: grad_norm 0.14551
|
| 207 |
+
wandb: lr 3e-05
|
| 208 |
+
wandb: mem_gb 22.02
|
| 209 |
+
wandb: step 150
|
| 210 |
+
wandb: t_data_s 0
|
| 211 |
+
wandb: +4 ...
|
| 212 |
+
wandb:
|
| 213 |
+
wandb: 🚀 View run glean-math-keep75-s1224 at: https://wandb.ai/hbfreed/glean-grid/runs/d06of2kq
|
| 214 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid
|
| 215 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 216 |
+
wandb: Find logs at: outputs/healed/grid_math/glean_keep75_s1224/wandb/run-20260716_151441-d06of2kq/logs
|
| 217 |
+
{
|
| 218 |
+
"correct": 905,
|
| 219 |
+
"accuracy": 0.686125852918878,
|
| 220 |
+
"finished": 1314,
|
| 221 |
+
"finish_rate": 0.9962092494313874,
|
| 222 |
+
"mean_completion_tokens": 113.11675511751326
|
| 223 |
+
}
|
| 224 |
+
saved item-level results -> outputs/evals/grid_math/glean_keep75_s1224_step100_chat.json
|
| 225 |
+
{
|
| 226 |
+
"correct": 912,
|
| 227 |
+
"accuracy": 0.6914329037149356,
|
| 228 |
+
"finished": 1314,
|
| 229 |
+
"finish_rate": 0.9962092494313874,
|
| 230 |
+
"mean_completion_tokens": 113.21000758150113
|
| 231 |
+
}
|
| 232 |
+
saved item-level results -> outputs/evals/grid_math/glean_keep75_s1224_step150_chat.json
|
healed/grid_math/glean_keep75_s1225.console.log
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 6 |
+
wandb: Run data is saved locally in outputs/healed/grid_math/glean_keep75_s1225/wandb/run-20260716_142818-9x4iij2d
|
| 7 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 8 |
+
wandb: Syncing run glean-math-keep75-s1225
|
| 9 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 10 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/9x4iij2d
|
| 11 |
+
|
| 12 |
+
resumed student weights from outputs/healed/grid_math/glean_keep75_s1225/step0100 (fresh optimizer, step counter at 0)
|
| 13 |
+
12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 5.31B | teacher overlap=False
|
| 14 |
+
restored optimizer/scheduler state from step 100; rebuilt 260 paged buffers
|
| 15 |
+
{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016109059700369834, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 63.3, "frames": {"chat": 223}, "mem_gb": 21.95}
|
| 16 |
+
{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016362527111552966, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.21875, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 206}, "mem_gb": 22.1}
|
| 17 |
+
{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.013859456993810212, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.228515625, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 213}, "mem_gb": 22.02}
|
| 18 |
+
{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018646482431787688, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 223}, "mem_gb": 21.96}
|
| 19 |
+
{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015217417630545484, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.201171875, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 227}, "mem_gb": 22.07}
|
| 20 |
+
{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018008409859096478, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 253}, "mem_gb": 22.09}
|
| 21 |
+
{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012912440449073135, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.1865234375, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 216}, "mem_gb": 22.1}
|
| 22 |
+
{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010693734145350754, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.1748046875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 205}, "mem_gb": 22.07}
|
| 23 |
+
{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015156030047448197, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 207}, "mem_gb": 22.16}
|
| 24 |
+
{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013672215391354015, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.1982421875, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 215}, "mem_gb": 22.08}
|
| 25 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 26 |
+
[eval step 110] sample: 'To solve this problem, we need to analyze the spiral pattern of numbers from 1 to 49 arranged on a square grid and identify the four shaded squares that lie on the same diagonal as the number 7. We th'
|
| 27 |
+
{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011061214675944453, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.2060546875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 228}, "mem_gb": 22.1}
|
| 28 |
+
{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012901123281742912, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.1982421875, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 221}, "mem_gb": 22.14}
|
| 29 |
+
{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009574843731914492, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.15625, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 254}, "mem_gb": 21.93}
|
| 30 |
+
{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011826626441131036, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.2392578125, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 210}, "mem_gb": 22.06}
|
| 31 |
+
{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010666260581545066, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.166015625, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 226}, "mem_gb": 22.02}
|
| 32 |
+
{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01181182961029117, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 212}, "mem_gb": 22.09}
|
| 33 |
+
{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014291961643429628, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.20703125, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.2, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 211}, "mem_gb": 22.02}
|
| 34 |
+
{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011632958819546426, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.1640625, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 196}, "mem_gb": 22.07}
|
| 35 |
+
{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011126866567115454, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.2119140625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 212}, "mem_gb": 22.09}
|
| 36 |
+
{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010413994966812121, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.16015625, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 244}, "mem_gb": 22.0}
|
| 37 |
+
[eval step 120] sample: "To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers on the same diagonal as the number 7. Let's break down the problem step-by-step:\n\n1. **Underst"
|
| 38 |
+
{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010520029380288906, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.166015625, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 222}, "mem_gb": 22.05}
|
| 39 |
+
{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011724793996859807, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.1884765625, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 218}, "mem_gb": 22.09}
|
| 40 |
+
{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011794728558894713, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.1982421875, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 252}, "mem_gb": 21.97}
|
| 41 |
+
{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012576708099556466, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.19140625, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 202}, "mem_gb": 22.14}
|
| 42 |
+
{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012525027039841128, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 237}, "mem_gb": 22.1}
|
| 43 |
+
{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011847056540916674, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.166015625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 234}, "mem_gb": 22.08}
|
| 44 |
+
{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010580201309620558, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 215}, "mem_gb": 22.1}
|
| 45 |
+
{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010746679509648433, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.1796875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 234}, "mem_gb": 22.03}
|
| 46 |
+
{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010033142010107016, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.1884765625, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 216}, "mem_gb": 22.08}
|
| 47 |
+
{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01059437686605379, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.1787109375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 210}, "mem_gb": 22.05}
|
| 48 |
+
[eval step 130] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid starting from the center. We then identify the four shaded squares that lie on the same diagonal'
|
| 49 |
+
{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011416423269287528, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.169921875, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 199}, "mem_gb": 22.09}
|
| 50 |
+
{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011045920558762736, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.1865234375, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 210}, "mem_gb": 22.11}
|
| 51 |
+
{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010189920482278103, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.1689453125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 225}, "mem_gb": 22.05}
|
| 52 |
+
{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011348977557197213, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.1728515625, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 253}, "mem_gb": 21.95}
|
| 53 |
+
{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010840521045137818, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.1513671875, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 247}, "mem_gb": 22.07}
|
| 54 |
+
{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01431039827777228, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.2021484375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 238}, "mem_gb": 22.07}
|
| 55 |
+
{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012493417616401954, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.16796875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 235}, "mem_gb": 22.09}
|
| 56 |
+
{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011775391096648916, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.1640625, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 215}, "mem_gb": 22.06}
|
| 57 |
+
{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011272327020104665, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.177734375, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 268}, "mem_gb": 22.06}
|
| 58 |
+
{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011195119415794033, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.1689453125, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 228}, "mem_gb": 22.09}
|
| 59 |
+
[eval step 140] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid and identify the four shaded squares that lie on the same diagonal as the number 7. Then, we wil'
|
| 60 |
+
{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01146084621019351, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.1513671875, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 252}, "mem_gb": 22.03}
|
| 61 |
+
{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01266345674659824, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.1875, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 223}, "mem_gb": 22.11}
|
| 62 |
+
{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012728927766492901, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.1689453125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 226}, "mem_gb": 22.09}
|
| 63 |
+
{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015653314464636303, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.2421875, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 208}, "mem_gb": 22.14}
|
| 64 |
+
{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.00990355064412579, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.154296875, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.6, "frames": {"chat": 240}, "mem_gb": 22.03}
|
| 65 |
+
{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011352584307268262, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.171875, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 222}, "mem_gb": 22.02}
|
| 66 |
+
{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009270905922904301, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.1396484375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 236}, "mem_gb": 22.09}
|
| 67 |
+
{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010650625815118353, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.181640625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 217}, "mem_gb": 22.06}
|
| 68 |
+
{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010312878351037702, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.1611328125, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 252}, "mem_gb": 21.97}
|
| 69 |
+
{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009814306650865667, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.14453125, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.7, "frames": {"chat": 222}, "mem_gb": 22.08}
|
| 70 |
+
[eval step 150] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid and identify the four numbers that lie on the same diagonal as the number 7. We then need to det'
|
| 71 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep75_s1225/step0150
|
| 72 |
+
wandb: updating run metadata
|
| 73 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 74 |
+
wandb:
|
| 75 |
+
wandb: Run history:
|
| 76 |
+
wandb: comp_len ▅▇▆▅▄▆▇▇▆▄▆▅▆▆█▃▅▅▂▇▄▆▄▆▆▆▅▂▃▃▆▁▂▅▅▃▅▄▅▅
|
| 77 |
+
wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇███
|
| 78 |
+
wandb: epoch ▁▁▁▁▁███████████████████████████████████
|
| 79 |
+
wandb: finish_rate ▃▃▃▅▅▃▂▁▄▂▅▅▄▂▃▆▅▃█▁▆▄▆▄▄▄▇█▇▅▄█▄▆▄▁▅▆▅▅
|
| 80 |
+
wandb: forward_topk_kl ▆▆▄█▅▄▂▅▄▂▁▃▂▃▅▂▂▃▃▃▃▂▂▂▂▂▂▃▂▅▃▂▂▃▄▁▃▁▂▁
|
| 81 |
+
wandb: grad_norm ▂▃▃▃▂��▂▃▂▃▁▂▂▂▃▂▂▂▂█▂▂▂▂▂▂▂▁▂▂▂▂▁▂▂▁▂▁▂▁
|
| 82 |
+
wandb: lr ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 83 |
+
wandb: mem_gb ▂▆▄▂▅▆▅█▆▆▁▅▄▅▆▅▆▂▇▆▆▄▆▅▆▅▂▅▅▆▅▆▄▆▆▄▄▆▅▆
|
| 84 |
+
wandb: step ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇███
|
| 85 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 86 |
+
wandb: +3 ...
|
| 87 |
+
wandb:
|
| 88 |
+
wandb: Run summary:
|
| 89 |
+
wandb: comp_len 540.5
|
| 90 |
+
wandb: cumulative_loss_tokens 18000000
|
| 91 |
+
wandb: epoch 2
|
| 92 |
+
wandb: finish_rate 0.847
|
| 93 |
+
wandb: forward_topk_kl 0.00981
|
| 94 |
+
wandb: grad_norm 0.14453
|
| 95 |
+
wandb: lr 3e-05
|
| 96 |
+
wandb: mem_gb 22.08
|
| 97 |
+
wandb: step 150
|
| 98 |
+
wandb: t_data_s 0
|
| 99 |
+
wandb: +4 ...
|
| 100 |
+
wandb:
|
| 101 |
+
wandb: 🚀 View run glean-math-keep75-s1225 at: https://wandb.ai/hbfreed/glean-grid/runs/9x4iij2d
|
| 102 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid
|
| 103 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 104 |
+
wandb: Find logs at: outputs/healed/grid_math/glean_keep75_s1225/wandb/run-20260716_142818-9x4iij2d/logs
|
| 105 |
+
{
|
| 106 |
+
"correct": 910,
|
| 107 |
+
"accuracy": 0.6899166034874905,
|
| 108 |
+
"finished": 1315,
|
| 109 |
+
"finish_rate": 0.9969673995451099,
|
| 110 |
+
"mean_completion_tokens": 113.55724033358605
|
| 111 |
+
}
|
| 112 |
+
saved item-level results -> outputs/evals/grid_math/glean_keep75_s1225_step100_chat.json
|
| 113 |
+
{
|
| 114 |
+
"correct": 909,
|
| 115 |
+
"accuracy": 0.689158453373768,
|
| 116 |
+
"finished": 1316,
|
| 117 |
+
"finish_rate": 0.9977255496588324,
|
| 118 |
+
"mean_completion_tokens": 114.47536012130402
|
| 119 |
+
}
|
| 120 |
+
saved item-level results -> outputs/evals/grid_math/glean_keep75_s1225_step150_chat.json
|
healed/grid_math/glean_keep75_s1226.console.log
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 6 |
+
wandb: Run data is saved locally in outputs/healed/grid_math/glean_keep75_s1226/wandb/run-20260716_142818-kru5sldj
|
| 7 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 8 |
+
wandb: Syncing run glean-math-keep75-s1226
|
| 9 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 10 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/kru5sldj
|
| 11 |
+
|
| 12 |
+
resumed student weights from outputs/healed/grid_math/glean_keep75_s1226/step0100 (fresh optimizer, step counter at 0)
|
| 13 |
+
12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 5.31B | teacher overlap=False
|
| 14 |
+
restored optimizer/scheduler state from step 100; rebuilt 260 paged buffers
|
| 15 |
+
{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017711289564648177, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 62.0, "frames": {"chat": 232}, "mem_gb": 21.89}
|
| 16 |
+
{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017982081608990362, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.832, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 220}, "mem_gb": 22.09}
|
| 17 |
+
{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0159690285191716, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.1982421875, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 210}, "mem_gb": 22.14}
|
| 18 |
+
{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.014078767126984894, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 226}, "mem_gb": 22.06}
|
| 19 |
+
{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015146795610602325, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.248046875, "lr": 3e-05, "finish_rate": 0.741, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 212}, "mem_gb": 22.09}
|
| 20 |
+
{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.014050660304430251, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.201171875, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 236}, "mem_gb": 22.1}
|
| 21 |
+
{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012824523648739948, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.928, "comp_len": 454.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 264}, "mem_gb": 21.97}
|
| 22 |
+
{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.017827476510805233, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.2421875, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 229}, "mem_gb": 22.07}
|
| 23 |
+
{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01010443297145733, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.16015625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 465.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 258}, "mem_gb": 21.95}
|
| 24 |
+
{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013931514149834403, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.19140625, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 208}, "mem_gb": 22.11}
|
| 25 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 26 |
+
[eval step 110] sample: 'To solve this problem, we need to understand the geometric properties involved. When the midpoints of the sides of a triangle are connected, the segments joining these midpoints form a smaller triangl'
|
| 27 |
+
{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010035461849397204, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.16015625, "lr": 3e-05, "finish_rate": 0.88, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 249}, "mem_gb": 22.02}
|
| 28 |
+
{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01025480254034434, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.1796875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 220}, "mem_gb": 22.09}
|
| 29 |
+
{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010523743329072991, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.1708984375, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 223}, "mem_gb": 22.08}
|
| 30 |
+
{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01153164215181023, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.17578125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 221}, "mem_gb": 22.09}
|
| 31 |
+
{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009408667080748516, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.1552734375, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 230}, "mem_gb": 21.99}
|
| 32 |
+
{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011410523822395286, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.1630859375, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 214}, "mem_gb": 22.07}
|
| 33 |
+
{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01616842130373698, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.224609375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 214}, "mem_gb": 22.08}
|
| 34 |
+
{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013691102667020944, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.19921875, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 210}, "mem_gb": 22.13}
|
| 35 |
+
{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013348402527160942, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.1884765625, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 214}, "mem_gb": 22.09}
|
| 36 |
+
{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011879650606094704, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.2001953125, "lr": 3e-05, "finish_rate": 0.791, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 215}, "mem_gb": 22.05}
|
| 37 |
+
[eval step 120] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected by segments. This process creates a new triangle, known as th'
|
| 38 |
+
{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013821079181631406, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.3, "frames": {"chat": 208}, "mem_gb": 22.09}
|
| 39 |
+
{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010628624554680815, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.1708984375, "lr": 3e-05, "finish_rate": 0.789, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 218}, "mem_gb": 21.97}
|
| 40 |
+
{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010161227444757242, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.1640625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 233}, "mem_gb": 21.99}
|
| 41 |
+
{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009278946231456938, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.16015625, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.3, "frames": {"chat": 231}, "mem_gb": 22.03}
|
| 42 |
+
{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011464818606327754, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.1611328125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 235}, "mem_gb": 22.22}
|
| 43 |
+
{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010276864761835895, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.1484375, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 216}, "mem_gb": 22.08}
|
| 44 |
+
{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009667191798787098, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.162109375, "lr": 3e-05, "finish_rate": 0.831, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 237}, "mem_gb": 22.1}
|
| 45 |
+
{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014880005238496233, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 203}, "mem_gb": 22.1}
|
| 46 |
+
{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010798121559165885, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.166015625, "lr": 3e-05, "finish_rate": 0.873, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 236}, "mem_gb": 22.13}
|
| 47 |
+
{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009982232672628015, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.1513671875, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 214}, "mem_gb": 22.1}
|
| 48 |
+
[eval step 130] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected by segments. This process creates a new triangle, known as th'
|
| 49 |
+
{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01276522857361318, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.1826171875, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 209}, "mem_gb": 22.09}
|
| 50 |
+
{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01355358442418122, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.2216796875, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.2, "frames": {"chat": 256}, "mem_gb": 22.1}
|
| 51 |
+
{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.00979031468022537, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.1591796875, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.9, "frames": {"chat": 230}, "mem_gb": 21.96}
|
| 52 |
+
{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012052156168699731, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 230}, "mem_gb": 22.15}
|
| 53 |
+
{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011693784629782506, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.1640625, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 227}, "mem_gb": 22.05}
|
| 54 |
+
{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011806925121663758, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.150390625, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.8, "frames": {"chat": 208}, "mem_gb": 22.11}
|
| 55 |
+
{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011330220682158445, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.166015625, "lr": 3e-05, "finish_rate": 0.699, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 206}, "mem_gb": 22.12}
|
| 56 |
+
{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010682711551792455, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.15625, "lr": 3e-05, "finish_rate": 0.82, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 228}, "mem_gb": 22.0}
|
| 57 |
+
{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011466035542547858, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.15234375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 224}, "mem_gb": 22.09}
|
| 58 |
+
{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012643014844014155, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.236328125, "lr": 3e-05, "finish_rate": 0.66, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 200}, "mem_gb": 22.13}
|
| 59 |
+
[eval step 140] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected by segments. This process creates a new triangle, known as th'
|
| 60 |
+
{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01080410843101951, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.1689453125, "lr": 3e-05, "finish_rate": 0.714, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 196}, "mem_gb": 22.11}
|
| 61 |
+
{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009731826097378507, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.140625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 223}, "mem_gb": 22.09}
|
| 62 |
+
{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011139519218046916, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.2119140625, "lr": 3e-05, "finish_rate": 0.869, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 213}, "mem_gb": 21.98}
|
| 63 |
+
{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009973561418584237, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.146484375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 232}, "mem_gb": 22.02}
|
| 64 |
+
{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009199376441648928, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.16015625, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 223}, "mem_gb": 22.02}
|
| 65 |
+
{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0111230931228473, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.1611328125, "lr": 3e-05, "finish_rate": 0.85, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 233}, "mem_gb": 22.11}
|
| 66 |
+
{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011176060675464882, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.1474609375, "lr": 3e-05, "finish_rate": 0.816, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 217}, "mem_gb": 22.11}
|
| 67 |
+
{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01491896515111827, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.752, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 202}, "mem_gb": 22.17}
|
| 68 |
+
{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.00973438565802838, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.1484375, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.5, "frames": {"chat": 253}, "mem_gb": 22.03}
|
| 69 |
+
{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.009605388223500147, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.1552734375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 231}, "mem_gb": 22.03}
|
| 70 |
+
[eval step 150] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected by segments. This process creates a new triangle, known as th'
|
| 71 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/glean_keep75_s1226/step0150
|
| 72 |
+
wandb: updating run metadata
|
| 73 |
+
wandb: uploading output.log
|
| 74 |
+
wandb:
|
| 75 |
+
wandb: Run history:
|
| 76 |
+
wandb: comp_len ▃▅▄▆▃▁▆▂▅▄▄▆▆▆▆▆▅▃▄▃▃▇▃▆▆▄▄▄▆▇▄▇█▄▆▄▃▅▇▄
|
| 77 |
+
wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇███
|
| 78 |
+
wandb: epoch ▁▁▁▁████████████████████████████████████
|
| 79 |
+
wandb: finish_rate ▆▅▄▅▃█▆▇▃▆▆▇▄▄▄▄▃▄▇▆▆▅▃▇▃▇▇▅▇▃▅▆▁▂▆▆▆▅▃▇
|
| 80 |
+
wandb: forward_topk_kl ██▆▅▆▄█▂▅▂▂▃▃▅▄▅▂▂▁▃▁▆▂▂▄▁▃▃▃▃▃▄▂▁▃▁▃▃▆▁
|
| 81 |
+
wandb: grad_norm ▆█▄▄▇▄▇▂▄▂▃▂▂▆▅▅▄▃▂▂▁▂█▃▂▆▂▄▂▂▂��▃▁▅▂▂▁▄▂
|
| 82 |
+
wandb: lr ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 83 |
+
wandb: mem_gb ▁▅▆▅▅▃▅▂▆▄▅▅▃▅▅▅▄▅▃▃█▅▅▅▆▅▂▇▄▆▃▅▆▆▅▄▄▆▆▄
|
| 84 |
+
wandb: step ▁▁▁▁▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇███
|
| 85 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 86 |
+
wandb: +3 ...
|
| 87 |
+
wandb:
|
| 88 |
+
wandb: Run summary:
|
| 89 |
+
wandb: comp_len 519.5
|
| 90 |
+
wandb: cumulative_loss_tokens 18000000
|
| 91 |
+
wandb: epoch 2
|
| 92 |
+
wandb: finish_rate 0.879
|
| 93 |
+
wandb: forward_topk_kl 0.00961
|
| 94 |
+
wandb: grad_norm 0.15527
|
| 95 |
+
wandb: lr 3e-05
|
| 96 |
+
wandb: mem_gb 22.03
|
| 97 |
+
wandb: step 150
|
| 98 |
+
wandb: t_data_s 0
|
| 99 |
+
wandb: +4 ...
|
| 100 |
+
wandb:
|
| 101 |
+
wandb: 🚀 View run glean-math-keep75-s1226 at: https://wandb.ai/hbfreed/glean-grid/runs/kru5sldj
|
| 102 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid
|
| 103 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 104 |
+
wandb: Find logs at: outputs/healed/grid_math/glean_keep75_s1226/wandb/run-20260716_142818-kru5sldj/logs
|
| 105 |
+
{
|
| 106 |
+
"correct": 915,
|
| 107 |
+
"accuracy": 0.6937073540561031,
|
| 108 |
+
"finished": 1314,
|
| 109 |
+
"finish_rate": 0.9962092494313874,
|
| 110 |
+
"mean_completion_tokens": 114.16527672479151
|
| 111 |
+
}
|
| 112 |
+
saved item-level results -> outputs/evals/grid_math/glean_keep75_s1226_step100_chat.json
|
| 113 |
+
{
|
| 114 |
+
"correct": 925,
|
| 115 |
+
"accuracy": 0.7012888551933283,
|
| 116 |
+
"finished": 1314,
|
| 117 |
+
"finish_rate": 0.9962092494313874,
|
| 118 |
+
"mean_completion_tokens": 115.14329037149355
|
| 119 |
+
}
|
| 120 |
+
saved item-level results -> outputs/evals/grid_math/glean_keep75_s1226_step150_chat.json
|
healed/grid_math/grid.log
ADDED
|
File without changes
|
healed/grid_math/reap_keep25_s1225.console.log
ADDED
|
@@ -0,0 +1,231 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: setting up run k6renipk
|
| 6 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 7 |
+
wandb: Run data is saved locally in outputs/healed/grid_math/reap_keep25_s1225/wandb/run-20260716_063505-k6renipk
|
| 8 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 9 |
+
wandb: Syncing run reap-math-keep25-s1225
|
| 10 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 11 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/k6renipk
|
| 12 |
+
12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 2.09B | teacher overlap=False
|
| 13 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 5.5052759854316715, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 91.0, "lr": 6e-06, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 191}, "mem_gb": 9.93}
|
| 14 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 15 |
+
[eval step 1] sample: '\n4,Christs up,\n\nTheadic<The,O(The,the part of theos:\n\nThe above, from the entire data in order,1, ForThe - ISS*Theemume, \\less numbered \n\nim. Botte, JBeva;'
|
| 16 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 5.34218070195516, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 72.5, "lr": 9e-06, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 219}, "mem_gb": 9.99}
|
| 17 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 5.2662361368576684, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 54.5, "lr": 1.2e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.4, "frames": {"chat": 207}, "mem_gb": 10.0}
|
| 18 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 4.323533807410796, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 49.75, "lr": 1.5e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.4, "frames": {"chat": 208}, "mem_gb": 9.95}
|
| 19 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 3.656937780322631, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 42.5, "lr": 1.8e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 219}, "mem_gb": 9.99}
|
| 20 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 3.098056636095047, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 32.25, "lr": 2.1e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 246}, "mem_gb": 9.87}
|
| 21 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 2.628579444358746, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 41.0, "lr": 2.4e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 206}, "mem_gb": 10.02}
|
| 22 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 2.0468826131433246, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 26.75, "lr": 2.7000000000000002e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 233}, "mem_gb": 10.0}
|
| 23 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.5576237986435493, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 16.125, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.7, "frames": {"chat": 229}, "mem_gb": 9.86}
|
| 24 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 1.2283661879966656, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 9.4375, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 236}, "mem_gb": 9.9}
|
| 25 |
+
[eval step 10] sample: 'To solve this problem, we need to determine the total number of prime numbers less than 49. We will first identify the prime numbers less than 49.\n\n1. **Ident the prime numbers less than 49:**\n - Th'
|
| 26 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.9977921682407459, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 6.375, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 239}, "mem_gb": 9.78}
|
| 27 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8474529584680994, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 6.25, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 241}, "mem_gb": 9.9}
|
| 28 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7915342311960956, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 15.375, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 226}, "mem_gb": 9.87}
|
| 29 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6635135169697305, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 3.984375, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.7, "frames": {"chat": 234}, "mem_gb": 10.0}
|
| 30 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6129945642106235, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 2.453125, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 257}, "mem_gb": 9.99}
|
| 31 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.714731409107397, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 3.296875, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 208}, "mem_gb": 10.04}
|
| 32 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5896475110622744, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 5.53125, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 211}, "mem_gb": 10.02}
|
| 33 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.517651187770317, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 1.390625, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 227}, "mem_gb": 10.0}
|
| 34 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5277626443808278, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 1.3984375, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 211}, "mem_gb": 9.98}
|
| 35 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4388222487750153, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 1.1875, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 238}, "mem_gb": 9.99}
|
| 36 |
+
[eval step 20] sample: "To solve this problem, we need to understand how spiral patterns and the arrangement of numbers on a square grid. Here's how we can break it down:\n\n1. **Understand the Spical Pattern:**\n The spiral "
|
| 37 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.42888357511336606, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 1.046875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 237}, "mem_gb": 10.03}
|
| 38 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.44955972477830947, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 1.171875, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 208}, "mem_gb": 10.03}
|
| 39 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.40363404167294503, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.9921875, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 221}, "mem_gb": 10.12}
|
| 40 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.36760696430206297, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.8828125, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 232}, "mem_gb": 9.96}
|
| 41 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3675604015878091, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.8515625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 208}, "mem_gb": 9.99}
|
| 42 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3380553434039156, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.7734375, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 227}, "mem_gb": 9.91}
|
| 43 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3542323801631729, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.7578125, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 221}, "mem_gb": 9.94}
|
| 44 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3140212787228326, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.7421875, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 232}, "mem_gb": 10.0}
|
| 45 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2983292136088014, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.71875, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 219}, "mem_gb": 10.0}
|
| 46 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3193080648737649, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.70703125, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.8, "frames": {"chat": 195}, "mem_gb": 10.09}
|
| 47 |
+
[eval step 30] sample: 'To solve this problem, we need to understand the pattern of the numbers in the spiral and identify the prime numbers that appear in the shaded squares.\n\n### Steps to Solve:\n\n1. **Identify the Pattern:'
|
| 48 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.31196173311385017, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.80859375, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.9, "frames": {"chat": 216}, "mem_gb": 10.0}
|
| 49 |
+
{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2535243794289728, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.60546875, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.9, "frames": {"chat": 208}, "mem_gb": 9.88}
|
| 50 |
+
{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.263915772453323, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.70703125, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 235}, "mem_gb": 9.88}
|
| 51 |
+
{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27100975296981633, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.7265625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.9, "frames": {"chat": 225}, "mem_gb": 9.99}
|
| 52 |
+
{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.34947378008762997, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.76953125, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 213}, "mem_gb": 10.08}
|
| 53 |
+
{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2621212104951342, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.71875, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 257}, "mem_gb": 9.75}
|
| 54 |
+
{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2858927722416818, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.6640625, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 212}, "mem_gb": 10.02}
|
| 55 |
+
{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24749040407985448, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 221}, "mem_gb": 10.0}
|
| 56 |
+
{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24993702379036695, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.6328125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.7, "frames": {"chat": 242}, "mem_gb": 9.99}
|
| 57 |
+
{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22095589282835523, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 220}, "mem_gb": 9.96}
|
| 58 |
+
[eval step 40] sample: 'To solve this problem, we need to understand the arrangement of numbers in a spiral pattern on a square grid and identify the prime numbers that appear in the shaded squares.\n\n### Steps to Solve the P'
|
| 59 |
+
{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24182025297402093, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.62890625, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 240}, "mem_gb": 9.85}
|
| 60 |
+
{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24479971672501416, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.60546875, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.6, "frames": {"chat": 206}, "mem_gb": 9.98}
|
| 61 |
+
{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2458774261167273, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 226}, "mem_gb": 10.0}
|
| 62 |
+
{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.31763079155124724, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 1.1640625, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 244}, "mem_gb": 9.78}
|
| 63 |
+
{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.26981097101569174, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.6, "frames": {"chat": 224}, "mem_gb": 10.0}
|
| 64 |
+
{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2382314425634841, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.625, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.1, "frames": {"chat": 271}, "mem_gb": 9.72}
|
| 65 |
+
{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21951183261151114, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.546875, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 236}, "mem_gb": 10.01}
|
| 66 |
+
{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24863433394009868, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.63671875, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 232}, "mem_gb": 9.88}
|
| 67 |
+
{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19559897173379237, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 210}, "mem_gb": 9.93}
|
| 68 |
+
{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18844997002581756, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.470703125, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.2, "frames": {"chat": 217}, "mem_gb": 9.9}
|
| 69 |
+
[eval step 50] sample: 'To solve this problem, we need to understand the arrangement of numbers in a spiral pattern on a square grid and identify the four numbers that will appear in the shaded squares, on the same diagonal '
|
| 70 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep25_s1225/step0050
|
| 71 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25687755517810584, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 224}, "mem_gb": 10.02}
|
| 72 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.28079532670788465, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 203}, "mem_gb": 9.87}
|
| 73 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2312026971814533, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.5546875, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 239}, "mem_gb": 9.97}
|
| 74 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19427155094649642, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 254}, "mem_gb": 9.88}
|
| 75 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18302697168818363, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.474609375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.5, "frames": {"chat": 241}, "mem_gb": 9.97}
|
| 76 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24095728847576925, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 213}, "mem_gb": 10.0}
|
| 77 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2061793549572428, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.74609375, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 221}, "mem_gb": 10.05}
|
| 78 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20495547246101, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.3, "frames": {"chat": 196}, "mem_gb": 10.01}
|
| 79 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17600186436952403, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 270}, "mem_gb": 9.81}
|
| 80 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15102526473104955, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 216}, "mem_gb": 9.99}
|
| 81 |
+
[eval step 60] sample: "To solve this problem, we need to determine how many of the four numbers that appear in the shaded squares on the square grid are prime. Here's a step-by-step approach to solve this problem:\n\n1. **Und"
|
| 82 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21443645971684405, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.57421875, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.6, "frames": {"chat": 200}, "mem_gb": 9.96}
|
| 83 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1430999824684734, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.9, "frames": {"chat": 206}, "mem_gb": 9.91}
|
| 84 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1475067599070569, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 234}, "mem_gb": 9.94}
|
| 85 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17827725302210698, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.439453125, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.3, "frames": {"chat": 215}, "mem_gb": 9.95}
|
| 86 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15364940941271682, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.494140625, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 255}, "mem_gb": 9.94}
|
| 87 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18826355347931384, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.546875, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 250}, "mem_gb": 9.82}
|
| 88 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17189633793290704, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 242}, "mem_gb": 9.99}
|
| 89 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2132726859041179, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 199}, "mem_gb": 10.0}
|
| 90 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.24157150672028463, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 208}, "mem_gb": 10.03}
|
| 91 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18809422614208113, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.3, "frames": {"chat": 208}, "mem_gb": 9.97}
|
| 92 |
+
[eval step 70] sample: 'To solve this problem, we need to understand how the spiral pattern works on a square grid and identify the numbers that will appear in the shaded squares on the same diagonal as the number \\(7\\).\n\n##'
|
| 93 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19989227577894927, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 209}, "mem_gb": 10.12}
|
| 94 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1725833379857863, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.484375, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 235}, "mem_gb": 9.95}
|
| 95 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1478453546665609, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.8, "frames": {"chat": 204}, "mem_gb": 9.94}
|
| 96 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21500386069975794, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 208}, "mem_gb": 10.0}
|
| 97 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1554392906052371, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 240}, "mem_gb": 10.0}
|
| 98 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1832204219336311, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 246}, "mem_gb": 9.99}
|
| 99 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17209678962721178, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 243}, "mem_gb": 9.81}
|
| 100 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20465335384116817, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.8, "frames": {"chat": 208}, "mem_gb": 10.01}
|
| 101 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1694449459930261, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.7, "frames": {"chat": 219}, "mem_gb": 10.0}
|
| 102 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.22224440845083446, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.5234375, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 211}, "mem_gb": 10.01}
|
| 103 |
+
[eval step 80] sample: 'To solve this problem, we need to understand the arrangement of numbers on a square grid and identify the numbers that will appear in the shaded squares, on the same diagonal as the number \\(7\\).\n\n###'
|
| 104 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1509838725623054, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 232}, "mem_gb": 9.97}
|
| 105 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17593013638195892, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 214}, "mem_gb": 10.0}
|
| 106 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15619046537938217, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 226}, "mem_gb": 9.89}
|
| 107 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17479738595858216, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 210}, "mem_gb": 10.01}
|
| 108 |
+
{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1404316657436701, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 218}, "mem_gb": 9.83}
|
| 109 |
+
{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1403924273949427, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 233}, "mem_gb": 9.98}
|
| 110 |
+
{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17982150458991528, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 215}, "mem_gb": 10.0}
|
| 111 |
+
{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18029257206898183, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 233}, "mem_gb": 9.99}
|
| 112 |
+
{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14494490633426854, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 209}, "mem_gb": 9.94}
|
| 113 |
+
{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15547993779405953, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 262}, "mem_gb": 9.87}
|
| 114 |
+
[eval step 90] sample: 'To solve this problem, we need to understand the arrangement of numbers on a square grid and identify the numbers that are prime and located on the same diagonal as the number \\(7\\).\n\n### Steps to Sol'
|
| 115 |
+
{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13846087997810294, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.384765625, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 249}, "mem_gb": 9.96}
|
| 116 |
+
{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.20217994603576758, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.4765625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 227}, "mem_gb": 9.99}
|
| 117 |
+
{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14583957826520005, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 221}, "mem_gb": 9.99}
|
| 118 |
+
{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1660599309977765, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 234}, "mem_gb": 10.01}
|
| 119 |
+
{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12514705213187263, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.6, "frames": {"chat": 213}, "mem_gb": 9.95}
|
| 120 |
+
{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12465972695561747, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.9, "frames": {"chat": 213}, "mem_gb": 9.89}
|
| 121 |
+
{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15661951549227038, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 234}, "mem_gb": 9.92}
|
| 122 |
+
{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.14978609310248867, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 222}, "mem_gb": 9.99}
|
| 123 |
+
{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16640961758115638, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 227}, "mem_gb": 10.0}
|
| 124 |
+
{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16042816521745795, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.427734375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 218}, "mem_gb": 10.04}
|
| 125 |
+
[eval step 100] sample: 'To solve this problem, we need to understand how the spiral pattern works on a square grid and identify the numbers that will appear in the shaded squares on the same diagonal as the number \\(7\\).\n\n##'
|
| 126 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep25_s1225/step0100
|
| 127 |
+
{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.17738757199452568, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 223}, "mem_gb": 10.01}
|
| 128 |
+
{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.18987772140322873, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 206}, "mem_gb": 10.0}
|
| 129 |
+
{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15001419565274068, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 213}, "mem_gb": 9.92}
|
| 130 |
+
{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.19792243093705425, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.451171875, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 223}, "mem_gb": 9.86}
|
| 131 |
+
{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15784500931380316, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 227}, "mem_gb": 9.97}
|
| 132 |
+
{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.15715095629387846, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 253}, "mem_gb": 10.0}
|
| 133 |
+
{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.16428898396513736, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.48828125, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.4, "frames": {"chat": 216}, "mem_gb": 10.0}
|
| 134 |
+
{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13419136733350653, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.2, "frames": {"chat": 205}, "mem_gb": 9.97}
|
| 135 |
+
{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.17433834038786591, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.7, "frames": {"chat": 207}, "mem_gb": 10.06}
|
| 136 |
+
{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14855496705075105, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 215}, "mem_gb": 9.98}
|
| 137 |
+
[eval step 110] sample: 'To solve this problem, we need to understand how the spiral pattern works on a square grid and identify the numbers that will appear in the shaded squares on the same diagonal as the number \\(7\\).\n\n##'
|
| 138 |
+
{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11281266368478536, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.43359375, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.7, "frames": {"chat": 228}, "mem_gb": 10.0}
|
| 139 |
+
{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14154021223733823, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.4453125, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 221}, "mem_gb": 10.04}
|
| 140 |
+
{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11087060506536314, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.2, "frames": {"chat": 254}, "mem_gb": 9.84}
|
| 141 |
+
{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09634949580542743, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 210}, "mem_gb": 9.96}
|
| 142 |
+
{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11305949703895797, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 226}, "mem_gb": 9.92}
|
| 143 |
+
{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11892387241683901, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 212}, "mem_gb": 9.99}
|
| 144 |
+
{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14203277846990775, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.0, "frames": {"chat": 211}, "mem_gb": 9.92}
|
| 145 |
+
{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1415941094346655, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.4, "frames": {"chat": 196}, "mem_gb": 9.97}
|
| 146 |
+
{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10169251247107362, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.5, "frames": {"chat": 212}, "mem_gb": 9.99}
|
| 147 |
+
{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10335419424610834, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.8, "frames": {"chat": 244}, "mem_gb": 9.91}
|
| 148 |
+
[eval step 120] sample: 'To solve this problem, we need to understand how the numbers are arranged in a spiral pattern on a square grid and identify the four numbers that are on the same diagonal as the number \\(7\\). Our goal'
|
| 149 |
+
{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1078583001211596, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 222}, "mem_gb": 9.95}
|
| 150 |
+
{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1223276870971856, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.0, "frames": {"chat": 218}, "mem_gb": 10.0}
|
| 151 |
+
{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12229978259069224, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.349609375, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 252}, "mem_gb": 9.87}
|
| 152 |
+
{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1320776972546242, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.3984375, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.2, "frames": {"chat": 202}, "mem_gb": 10.05}
|
| 153 |
+
{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1445485326328315, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.8, "frames": {"chat": 237}, "mem_gb": 10.0}
|
| 154 |
+
{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1339294173414198, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 234}, "mem_gb": 9.98}
|
| 155 |
+
{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09875204861378298, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 215}, "mem_gb": 10.0}
|
| 156 |
+
{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10042380276505525, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 234}, "mem_gb": 9.93}
|
| 157 |
+
{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0915526939183784, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 216}, "mem_gb": 9.98}
|
| 158 |
+
{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10634777345014736, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.7, "frames": {"chat": 210}, "mem_gb": 9.95}
|
| 159 |
+
[eval step 130] sample: 'To solve this problem, we need to understand the structure of the spiral pattern on the square grid and identify the numbers that will appear in the shaded squares on the same diagonal as the number \\'
|
| 160 |
+
{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12015755086155452, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.8, "frames": {"chat": 199}, "mem_gb": 10.0}
|
| 161 |
+
{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1101692976130483, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.7, "frames": {"chat": 210}, "mem_gb": 10.01}
|
| 162 |
+
{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.11059571424775447, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 225}, "mem_gb": 9.95}
|
| 163 |
+
{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10813538532971094, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 253}, "mem_gb": 9.85}
|
| 164 |
+
{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12084263433534652, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 247}, "mem_gb": 9.97}
|
| 165 |
+
{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13201027247390398, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.388671875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 238}, "mem_gb": 9.97}
|
| 166 |
+
{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13818833304295938, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.4, "frames": {"chat": 235}, "mem_gb": 9.99}
|
| 167 |
+
{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1345420230248322, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 215}, "mem_gb": 9.96}
|
| 168 |
+
{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10943617184103156, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 268}, "mem_gb": 9.97}
|
| 169 |
+
{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12807247791942208, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 228}, "mem_gb": 10.0}
|
| 170 |
+
[eval step 140] sample: 'To solve this problem, we need to understand how the spiral pattern works on a square grid and identify the numbers that will appear in the shaded squares on the same diagonal as the number \\(7\\).\n\n##'
|
| 171 |
+
{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12706402302297454, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.38671875, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 252}, "mem_gb": 9.93}
|
| 172 |
+
{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12039303982133667, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.1, "frames": {"chat": 223}, "mem_gb": 10.01}
|
| 173 |
+
{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.14205731437917177, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.7, "frames": {"chat": 226}, "mem_gb": 9.99}
|
| 174 |
+
{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1410405628043848, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 208}, "mem_gb": 10.04}
|
| 175 |
+
{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.09851235140593101, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 240}, "mem_gb": 9.93}
|
| 176 |
+
{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.13880995498470342, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.396484375, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.3, "frames": {"chat": 222}, "mem_gb": 9.92}
|
| 177 |
+
{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10342208843190843, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.5, "frames": {"chat": 236}, "mem_gb": 9.99}
|
| 178 |
+
{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.1016613077900062, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.6, "frames": {"chat": 217}, "mem_gb": 9.96}
|
| 179 |
+
{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.12922203347664327, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.9, "frames": {"chat": 252}, "mem_gb": 9.87}
|
| 180 |
+
{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.10404405277933304, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 222}, "mem_gb": 9.99}
|
| 181 |
+
[eval step 150] sample: 'To solve this problem, we need to understand how the spiral pattern works on a square grid and identify the numbers that will appear on the same diagonal as the number \\(7\\).\n\n### Steps to Solve the P'
|
| 182 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep25_s1225/step0150
|
| 183 |
+
wandb: updating run metadata
|
| 184 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 185 |
+
wandb:
|
| 186 |
+
wandb: Run history:
|
| 187 |
+
wandb: comp_len ▆▂▂▆▆▄▃█▆▆▆▂▅█▅▃▆▆▂▆▃▅▁▅▅▄▄▅▅▁▆▄▃▃▃▆▁▄▄▂
|
| 188 |
+
wandb: cumulative_loss_tokens ▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▄▅▆▆▆▆▆▆▆▆▆▇▇▇▇▇▇▇███
|
| 189 |
+
wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅████████
|
| 190 |
+
wandb: finish_rate ▅▃▆█▄▅█▆▃█▅▃▇▆█▇▂▇▄▆▃▃▇▅▄▃▂▇▅▆▁█▅▄▆▄▇▅▆▅
|
| 191 |
+
wandb: forward_topk_kl █▅▃▃▃▂▂▂▂▁▂▁▁▂▁▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 192 |
+
wandb: grad_norm █▆▃▂▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 193 |
+
wandb: lr ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 194 |
+
wandb: mem_gb ▆▆▆▄▆▆▇▃▆▁▃▅▆▃▅▆▆▅▅▂▆█▆▆▅▃▅▃▄▆▅▆▅▆▄▅▅▃▅▃
|
| 195 |
+
wandb: step ▁▁▁▁▁▂▂▂▂▂▂▃▃▃▃▄▄▄▄▄▄▄▅▅▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇█
|
| 196 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 197 |
+
wandb: +3 ...
|
| 198 |
+
wandb:
|
| 199 |
+
wandb: Run summary:
|
| 200 |
+
wandb: comp_len 540.5
|
| 201 |
+
wandb: cumulative_loss_tokens 18000000
|
| 202 |
+
wandb: epoch 2
|
| 203 |
+
wandb: finish_rate 0.847
|
| 204 |
+
wandb: forward_topk_kl 0.10404
|
| 205 |
+
wandb: grad_norm 0.35938
|
| 206 |
+
wandb: lr 3e-05
|
| 207 |
+
wandb: mem_gb 9.99
|
| 208 |
+
wandb: step 150
|
| 209 |
+
wandb: t_data_s 0
|
| 210 |
+
wandb: +4 ...
|
| 211 |
+
wandb:
|
| 212 |
+
wandb: 🚀 View run reap-math-keep25-s1225 at: https://wandb.ai/hbfreed/glean-grid/runs/k6renipk
|
| 213 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid
|
| 214 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 215 |
+
wandb: Find logs at: outputs/healed/grid_math/reap_keep25_s1225/wandb/run-20260716_063505-k6renipk/logs
|
| 216 |
+
{
|
| 217 |
+
"correct": 146,
|
| 218 |
+
"accuracy": 0.1106899166034875,
|
| 219 |
+
"finished": 1059,
|
| 220 |
+
"finish_rate": 0.8028809704321456,
|
| 221 |
+
"mean_completion_tokens": 204.20015163002273
|
| 222 |
+
}
|
| 223 |
+
saved item-level results -> outputs/evals/grid_math/reap_keep25_s1225_step100_chat.json
|
| 224 |
+
{
|
| 225 |
+
"correct": 178,
|
| 226 |
+
"accuracy": 0.13495072024260804,
|
| 227 |
+
"finished": 1079,
|
| 228 |
+
"finish_rate": 0.8180439727065959,
|
| 229 |
+
"mean_completion_tokens": 200.11675511751326
|
| 230 |
+
}
|
| 231 |
+
saved item-level results -> outputs/evals/grid_math/reap_keep25_s1225_step150_chat.json
|
healed/grid_math/reap_keep75_s1225.console.log
ADDED
|
@@ -0,0 +1,231 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: setting up run y9ttnb1t
|
| 6 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 7 |
+
wandb: Run data is saved locally in outputs/healed/grid_math/reap_keep75_s1225/wandb/run-20260716_172809-y9ttnb1t
|
| 8 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 9 |
+
wandb: Syncing run reap-math-keep75-s1225
|
| 10 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 11 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/y9ttnb1t
|
| 12 |
+
|
| 13 |
+
12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 5.31B | teacher overlap=False
|
| 14 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.10482486505489796, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 2.21875, "lr": 6e-06, "finish_rate": 0.733, "comp_len": 628.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 191}, "mem_gb": 21.94}
|
| 15 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 16 |
+
[eval step 1] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid and identify the four shaded squares that lie on the same diagonal as the number 7. We then need'
|
| 17 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.09455605004470174, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 1.7421875, "lr": 9e-06, "finish_rate": 0.845, "comp_len": 547.9, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 219}, "mem_gb": 22.09}
|
| 18 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.11970494873368492, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 2.234375, "lr": 1.2e-05, "finish_rate": 0.778, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 207}, "mem_gb": 22.1}
|
| 19 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1343396064637229, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 1.5234375, "lr": 1.5e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 208}, "mem_gb": 22.05}
|
| 20 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07346981081624204, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 1.0234375, "lr": 1.8e-05, "finish_rate": 0.799, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 219}, "mem_gb": 22.09}
|
| 21 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05854326014304534, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 0.83984375, "lr": 2.1e-05, "finish_rate": 0.915, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 246}, "mem_gb": 21.97}
|
| 22 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06574682900101567, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 1.0, "lr": 2.4e-05, "finish_rate": 0.704, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 206}, "mem_gb": 22.12}
|
| 23 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.055566156748325254, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 0.58203125, "lr": 2.7000000000000002e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 233}, "mem_gb": 22.1}
|
| 24 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06972100379186061, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 0.78125, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 229}, "mem_gb": 21.96}
|
| 25 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.049855469060580554, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 0.51953125, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 236}, "mem_gb": 22.0}
|
| 26 |
+
[eval step 10] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid and identify the four shaded squares that lie on the same diagonal as the number 7. Then, we nee'
|
| 27 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05346712346045921, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 0.466796875, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 239}, "mem_gb": 21.88}
|
| 28 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04576633981382474, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 241}, "mem_gb": 22.0}
|
| 29 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04779292347819234, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 226}, "mem_gb": 21.97}
|
| 30 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04039918827950023, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.41015625, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 234}, "mem_gb": 22.1}
|
| 31 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03985718289677364, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.1, "frames": {"chat": 257}, "mem_gb": 22.09}
|
| 32 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05252557928330886, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 208}, "mem_gb": 22.14}
|
| 33 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05430897771349798, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.4140625, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 211}, "mem_gb": 22.12}
|
| 34 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04499013489703648, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 227}, "mem_gb": 22.1}
|
| 35 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04699158584792167, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 211}, "mem_gb": 22.08}
|
| 36 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.039192986356653275, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.3, "frames": {"chat": 238}, "mem_gb": 22.09}
|
| 37 |
+
[eval step 20] sample: 'To solve this problem, we need to understand the spiral pattern of numbers from 1 to 49 arranged on a square grid. The spiral pattern typically follows a clockwise or counterclockwise direction, and w'
|
| 38 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.037524048287545644, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.1, "frames": {"chat": 237}, "mem_gb": 22.13}
|
| 39 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.046145012051596615, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 208}, "mem_gb": 22.13}
|
| 40 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.037229343083966526, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 221}, "mem_gb": 22.22}
|
| 41 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03738942670936231, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 232}, "mem_gb": 22.06}
|
| 42 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.041476121828503285, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 208}, "mem_gb": 22.09}
|
| 43 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.031991650097545545, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 227}, "mem_gb": 22.01}
|
| 44 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0368176663079299, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 221}, "mem_gb": 22.04}
|
| 45 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.035077836543767865, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 232}, "mem_gb": 22.1}
|
| 46 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.036738441793752526, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 219}, "mem_gb": 22.1}
|
| 47 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03953278637607582, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 195}, "mem_gb": 22.19}
|
| 48 |
+
[eval step 30] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid starting from the center. The goal is to identify the four numbers that lie on the same diagonal'
|
| 49 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.034143407355931896, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 216}, "mem_gb": 22.1}
|
| 50 |
+
{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03388264184119956, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 208}, "mem_gb": 21.98}
|
| 51 |
+
{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02752191697806896, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 235}, "mem_gb": 21.98}
|
| 52 |
+
{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02800570071142477, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 225}, "mem_gb": 22.09}
|
| 53 |
+
{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04181307503717641, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.5, "frames": {"chat": 213}, "mem_gb": 22.18}
|
| 54 |
+
{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.028032738463099425, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.2, "frames": {"chat": 257}, "mem_gb": 21.85}
|
| 55 |
+
{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03889315185185212, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 212}, "mem_gb": 22.12}
|
| 56 |
+
{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.033087143982380315, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.3203125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.1, "frames": {"chat": 221}, "mem_gb": 22.1}
|
| 57 |
+
{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.028922305932377153, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 242}, "mem_gb": 22.09}
|
| 58 |
+
{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03329299113704668, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 220}, "mem_gb": 22.06}
|
| 59 |
+
[eval step 40] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid starting from the center. The goal is to identify the four numbers that lie on the same diagonal'
|
| 60 |
+
{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.026630324330576695, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 240}, "mem_gb": 21.95}
|
| 61 |
+
{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.031542038596747445, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 206}, "mem_gb": 22.08}
|
| 62 |
+
{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04046216273989218, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 226}, "mem_gb": 22.1}
|
| 63 |
+
{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.049778220711982187, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.0, "frames": {"chat": 244}, "mem_gb": 21.88}
|
| 64 |
+
{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.032646800012839955, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 224}, "mem_gb": 22.1}
|
| 65 |
+
{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02689050944719929, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.5, "frames": {"chat": 271}, "mem_gb": 21.82}
|
| 66 |
+
{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.029499899335368538, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.3046875, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 236}, "mem_gb": 22.11}
|
| 67 |
+
{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03363814867006537, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.1, "frames": {"chat": 232}, "mem_gb": 21.98}
|
| 68 |
+
{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.030904324279655702, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 210}, "mem_gb": 22.03}
|
| 69 |
+
{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03020962130090532, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 217}, "mem_gb": 22.0}
|
| 70 |
+
[eval step 50] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid starting from the center. The numbers on the same diagonal as the number 7 need to be identified'
|
| 71 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep75_s1225/step0050
|
| 72 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03183544763289392, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 224}, "mem_gb": 22.12}
|
| 73 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.037371784832623475, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 203}, "mem_gb": 21.97}
|
| 74 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02824682667325251, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 239}, "mem_gb": 22.07}
|
| 75 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017269016665616074, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 254}, "mem_gb": 21.98}
|
| 76 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02183063491246042, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 241}, "mem_gb": 22.07}
|
| 77 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.023734719909347283, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.2265625, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 213}, "mem_gb": 22.1}
|
| 78 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.022771949865855276, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 221}, "mem_gb": 22.15}
|
| 79 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.024181459689225693, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 196}, "mem_gb": 22.11}
|
| 80 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015680761117596802, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.189453125, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.1, "frames": {"chat": 270}, "mem_gb": 21.91}
|
| 81 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01974021829657722, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.2197265625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 216}, "mem_gb": 22.09}
|
| 82 |
+
[eval step 60] sample: 'To solve this problem, we need to understand the spiral pattern of numbers from 1 to 49 arranged on a square grid and identify the four numbers that lie on the same diagonal as the number 7. Then, we '
|
| 83 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02411085371503917, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.244140625, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 200}, "mem_gb": 22.06}
|
| 84 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019210168254119343, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.23046875, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 206}, "mem_gb": 22.01}
|
| 85 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015362915280535041, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.1806640625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 234}, "mem_gb": 22.04}
|
| 86 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017122933285869658, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.1787109375, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 215}, "mem_gb": 22.05}
|
| 87 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01665975872905304, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.9, "frames": {"chat": 255}, "mem_gb": 22.04}
|
| 88 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01967068623775461, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.2490234375, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.0, "frames": {"chat": 250}, "mem_gb": 21.92}
|
| 89 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016327875597200666, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.1943359375, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 242}, "mem_gb": 22.09}
|
| 90 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.022421126494106526, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 199}, "mem_gb": 22.1}
|
| 91 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03170503772977584, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.283203125, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.1, "frames": {"chat": 208}, "mem_gb": 22.13}
|
| 92 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.024974758357469304, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 208}, "mem_gb": 22.07}
|
| 93 |
+
[eval step 70] sample: 'To solve this problem, we need to understand the spiral pattern of numbers from 1 to 49 arranged on a square grid and identify the four shaded squares that lie on the same diagonal as the number 7. Th'
|
| 94 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.023421097805319976, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.9, "frames": {"chat": 209}, "mem_gb": 22.22}
|
| 95 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01749415912704232, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.20703125, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 235}, "mem_gb": 22.05}
|
| 96 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019200553790223784, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 204}, "mem_gb": 22.04}
|
| 97 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.024120595507960144, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.9, "frames": {"chat": 208}, "mem_gb": 22.1}
|
| 98 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018558861436434867, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.2119140625, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 240}, "mem_gb": 22.1}
|
| 99 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.020433550003372755, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.2373046875, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.2, "frames": {"chat": 246}, "mem_gb": 22.09}
|
| 100 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019781218853243628, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.2109375, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 243}, "mem_gb": 21.91}
|
| 101 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0259718717401692, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 208}, "mem_gb": 22.11}
|
| 102 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.025042294318275527, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.2412109375, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 219}, "mem_gb": 22.1}
|
| 103 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.022973834389455928, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 211}, "mem_gb": 22.11}
|
| 104 |
+
[eval step 80] sample: "To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers on the same diagonal as the number 7. Let's break down the problem step-by-step:\n\n1. **Underst"
|
| 105 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.023045262621793274, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 232}, "mem_gb": 22.07}
|
| 106 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02766423754969922, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 214}, "mem_gb": 22.1}
|
| 107 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019795779521996154, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.2080078125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 226}, "mem_gb": 21.99}
|
| 108 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019442533940022502, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.2197265625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 210}, "mem_gb": 22.11}
|
| 109 |
+
{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018724862659955398, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 218}, "mem_gb": 21.93}
|
| 110 |
+
{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017705042916691553, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.197265625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 233}, "mem_gb": 22.08}
|
| 111 |
+
{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.021475131880197052, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.2177734375, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.9, "frames": {"chat": 215}, "mem_gb": 22.1}
|
| 112 |
+
{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018688833506122077, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.1, "frames": {"chat": 233}, "mem_gb": 22.09}
|
| 113 |
+
{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019814053858964082, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 209}, "mem_gb": 22.04}
|
| 114 |
+
{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015184357244963758, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.201171875, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.1, "frames": {"chat": 262}, "mem_gb": 21.97}
|
| 115 |
+
[eval step 90] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid starting from the center. The specific diagonal we are interested in is the one containing the n'
|
| 116 |
+
{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016755351828221077, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.1826171875, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.6, "frames": {"chat": 249}, "mem_gb": 22.06}
|
| 117 |
+
{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02191826228235538, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.224609375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.6, "frames": {"chat": 227}, "mem_gb": 22.09}
|
| 118 |
+
{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019205591833343107, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.2158203125, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 221}, "mem_gb": 22.09}
|
| 119 |
+
{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017255156058092445, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.197265625, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.3, "frames": {"chat": 234}, "mem_gb": 22.11}
|
| 120 |
+
{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017789849875991543, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.2119140625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 213}, "mem_gb": 22.05}
|
| 121 |
+
{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016172622948054535, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.1865234375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 213}, "mem_gb": 21.99}
|
| 122 |
+
{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016838261510718926, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.8, "frames": {"chat": 234}, "mem_gb": 22.02}
|
| 123 |
+
{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018477893968485294, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 222}, "mem_gb": 22.09}
|
| 124 |
+
{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018487704083664964, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.1, "frames": {"chat": 227}, "mem_gb": 22.1}
|
| 125 |
+
{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.021436278094242638, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.0, "frames": {"chat": 218}, "mem_gb": 22.14}
|
| 126 |
+
[eval step 100] sample: 'To solve this problem, we need to understand the spiral pattern of numbers from 1 to 49 arranged on a square grid. The spiral pattern typically follows a clockwise or counterclockwise direction, and w'
|
| 127 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep75_s1225/step0100
|
| 128 |
+
{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019779636907581395, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 223}, "mem_gb": 22.11}
|
| 129 |
+
{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01889594631587, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 206}, "mem_gb": 22.1}
|
| 130 |
+
{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01582208941851277, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.2236328125, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.2, "frames": {"chat": 213}, "mem_gb": 22.02}
|
| 131 |
+
{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02243306068785799, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.216796875, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 223}, "mem_gb": 21.96}
|
| 132 |
+
{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017156738052901348, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 227}, "mem_gb": 22.07}
|
| 133 |
+
{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017617856540104063, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.21484375, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.7, "frames": {"chat": 253}, "mem_gb": 22.1}
|
| 134 |
+
{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.018666838803018133, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.19921875, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 216}, "mem_gb": 22.1}
|
| 135 |
+
{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012518067852900518, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.15625, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 205}, "mem_gb": 22.07}
|
| 136 |
+
{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.016718426850946464, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.4, "frames": {"chat": 207}, "mem_gb": 22.16}
|
| 137 |
+
{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015750345862063114, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.1875, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.1, "frames": {"chat": 215}, "mem_gb": 22.08}
|
| 138 |
+
[eval step 110] sample: 'To solve this problem, we need to arrange the numbers from 1 to 49 in a spiral pattern on a square grid starting from the center. The specific diagonal we are interested in is the one containing the n'
|
| 139 |
+
{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01329283342436344, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 228}, "mem_gb": 22.1}
|
| 140 |
+
{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014201756338018458, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.1923828125, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 221}, "mem_gb": 22.14}
|
| 141 |
+
{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011235470508458092, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.146484375, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.7, "frames": {"chat": 254}, "mem_gb": 21.93}
|
| 142 |
+
{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013301954813903042, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 210}, "mem_gb": 22.06}
|
| 143 |
+
{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01213023890738453, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.15234375, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 226}, "mem_gb": 22.02}
|
| 144 |
+
{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014490734279165432, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.1669921875, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 212}, "mem_gb": 22.09}
|
| 145 |
+
{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015695976445505706, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.169921875, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.7, "frames": {"chat": 211}, "mem_gb": 22.02}
|
| 146 |
+
{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014398817314657694, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.1650390625, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 196}, "mem_gb": 22.07}
|
| 147 |
+
{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013426665164541919, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.1708984375, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 212}, "mem_gb": 22.09}
|
| 148 |
+
{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012419727628868229, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.16015625, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.1, "frames": {"chat": 244}, "mem_gb": 22.0}
|
| 149 |
+
[eval step 120] sample: "To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers on the same diagonal as the number 7. Let's break down the problem step-by-step:\n\n1. **Underst"
|
| 150 |
+
{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01267548363597598, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.1484375, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 222}, "mem_gb": 22.05}
|
| 151 |
+
{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012876643150672317, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.181640625, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 218}, "mem_gb": 22.1}
|
| 152 |
+
{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012929756409647719, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.1484375, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.2, "frames": {"chat": 252}, "mem_gb": 21.97}
|
| 153 |
+
{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014637890328870466, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.16796875, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 202}, "mem_gb": 22.15}
|
| 154 |
+
{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01425811582073026, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.1474609375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 237}, "mem_gb": 22.1}
|
| 155 |
+
{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01373342850942087, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.1513671875, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.7, "frames": {"chat": 234}, "mem_gb": 22.08}
|
| 156 |
+
{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0128611738221293, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.189453125, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.9, "frames": {"chat": 215}, "mem_gb": 22.1}
|
| 157 |
+
{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011419165648100898, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.1552734375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 234}, "mem_gb": 22.03}
|
| 158 |
+
{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01229629772514648, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.1689453125, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 216}, "mem_gb": 22.08}
|
| 159 |
+
{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012923938088014257, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.1708984375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.3, "frames": {"chat": 210}, "mem_gb": 22.05}
|
| 160 |
+
[eval step 130] sample: "To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers on the same diagonal as the number 7. Let's break down the problem step-by-step:\n\n1. **Underst"
|
| 161 |
+
{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01371961145355793, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.1533203125, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 199}, "mem_gb": 22.1}
|
| 162 |
+
{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012322454417881091, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.150390625, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 210}, "mem_gb": 22.11}
|
| 163 |
+
{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012123031584584776, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.1533203125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 225}, "mem_gb": 22.05}
|
| 164 |
+
{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012683820386894513, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.1533203125, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.7, "frames": {"chat": 253}, "mem_gb": 21.95}
|
| 165 |
+
{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011224880664504598, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.1337890625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 247}, "mem_gb": 22.07}
|
| 166 |
+
{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012574833552461738, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.1474609375, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 238}, "mem_gb": 22.07}
|
| 167 |
+
{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013766566757302886, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.15234375, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.0, "frames": {"chat": 235}, "mem_gb": 22.09}
|
| 168 |
+
{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013896340553753543, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.162109375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.3, "frames": {"chat": 215}, "mem_gb": 22.06}
|
| 169 |
+
{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014546819335536566, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.1904296875, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.0, "frames": {"chat": 268}, "mem_gb": 22.07}
|
| 170 |
+
{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012078923544633047, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.1552734375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 228}, "mem_gb": 22.1}
|
| 171 |
+
[eval step 140] sample: 'To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers on the same diagonal as the number 7. Then, we will determine which of these numbers are prime'
|
| 172 |
+
{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012921454821503722, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.1416015625, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.0, "frames": {"chat": 252}, "mem_gb": 22.03}
|
| 173 |
+
{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013791506547275154, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.166015625, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 223}, "mem_gb": 22.11}
|
| 174 |
+
{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01426673975478625, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.154296875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 226}, "mem_gb": 22.09}
|
| 175 |
+
{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.016324480891693385, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.3, "frames": {"chat": 208}, "mem_gb": 22.14}
|
| 176 |
+
{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011662282170122489, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.142578125, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.7, "frames": {"chat": 240}, "mem_gb": 22.03}
|
| 177 |
+
{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012903934089028431, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.169921875, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.0, "frames": {"chat": 222}, "mem_gb": 22.02}
|
| 178 |
+
{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.010760747992309432, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.142578125, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.5, "frames": {"chat": 236}, "mem_gb": 22.09}
|
| 179 |
+
{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013423855868071162, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 217}, "mem_gb": 22.06}
|
| 180 |
+
{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012151869152404834, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.1591796875, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.2, "frames": {"chat": 252}, "mem_gb": 21.97}
|
| 181 |
+
{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01206249614340874, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.1416015625, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 222}, "mem_gb": 22.09}
|
| 182 |
+
[eval step 150] sample: "To solve this problem, we need to understand the structure of the spiral pattern and identify the numbers on the same diagonal as the number 7. Let's break down the problem step-by-step:\n\n1. **Underst"
|
| 183 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep75_s1225/step0150
|
| 184 |
+
wandb: updating run metadata
|
| 185 |
+
wandb:
|
| 186 |
+
wandb: Run history:
|
| 187 |
+
wandb: comp_len █▆▄▂▄▄▆▅▄▆▅▃▃▅▇▆▅▁▅▄▃▆▂▅▄▆▂▆▅▂▃▅▅▃▆▄▂▅▄▅
|
| 188 |
+
wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▂▂▂▂▂▃▃▃▃▃▃▃▄▄▄▅▅▅▆▆▆▆▇▇▇▇▇▇████
|
| 189 |
+
wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅██████████████
|
| 190 |
+
wandb: finish_rate ▂▆▆▆▇▅▅▅▇▆█▅▃▅▂▂▂▃▄▆▇▅▇▅▅▃▄▇▃▆█▄▄▄█▆▆▄▁█
|
| 191 |
+
wandb: forward_topk_kl ▆█▃▄▃▃▃▃▂▃▂▂▂▃▂▁▁▁▂▁▂▂▂▂▂▁▁▁▁▂▁▁▁▁▁▁▁▁▁▁
|
| 192 |
+
wandb: grad_norm ▆▆▅▅▄▃▃▄▃▄▃▃▃▂▃▃▃▃▃▂▂▂▂▂▂▂▂▂▁▂▁▁▁▁▁▁▁▁█▁
|
| 193 |
+
wandb: lr ▁▂▃█████████████████████████████████████
|
| 194 |
+
wandb: mem_gb ▆▁▆▅▆▅▆▆▇▇▅▂▆▆▃▃▃▆▇▅▆█▅▅▆▅▆▇▆▄▃▅▆▅▅▅▅▅▆▅
|
| 195 |
+
wandb: step ▁▁▁▁▂▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇▇██
|
| 196 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 197 |
+
wandb: +3 ...
|
| 198 |
+
wandb:
|
| 199 |
+
wandb: Run summary:
|
| 200 |
+
wandb: comp_len 540.5
|
| 201 |
+
wandb: cumulative_loss_tokens 18000000
|
| 202 |
+
wandb: epoch 2
|
| 203 |
+
wandb: finish_rate 0.847
|
| 204 |
+
wandb: forward_topk_kl 0.01206
|
| 205 |
+
wandb: grad_norm 0.1416
|
| 206 |
+
wandb: lr 3e-05
|
| 207 |
+
wandb: mem_gb 22.09
|
| 208 |
+
wandb: step 150
|
| 209 |
+
wandb: t_data_s 0
|
| 210 |
+
wandb: +4 ...
|
| 211 |
+
wandb:
|
| 212 |
+
wandb: 🚀 View run reap-math-keep75-s1225 at: https://wandb.ai/hbfreed/glean-grid/runs/y9ttnb1t
|
| 213 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid
|
| 214 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 215 |
+
wandb: Find logs at: outputs/healed/grid_math/reap_keep75_s1225/wandb/run-20260716_172809-y9ttnb1t/logs
|
| 216 |
+
{
|
| 217 |
+
"correct": 894,
|
| 218 |
+
"accuracy": 0.6777862016679302,
|
| 219 |
+
"finished": 1314,
|
| 220 |
+
"finish_rate": 0.9962092494313874,
|
| 221 |
+
"mean_completion_tokens": 110.35633055344958
|
| 222 |
+
}
|
| 223 |
+
saved item-level results -> outputs/evals/grid_math/reap_keep75_s1225_step100_chat.json
|
| 224 |
+
{
|
| 225 |
+
"correct": 891,
|
| 226 |
+
"accuracy": 0.6755117513267627,
|
| 227 |
+
"finished": 1313,
|
| 228 |
+
"finish_rate": 0.9954510993176648,
|
| 229 |
+
"mean_completion_tokens": 111.60500379075057
|
| 230 |
+
}
|
| 231 |
+
saved item-level results -> outputs/evals/grid_math/reap_keep75_s1225_step150_chat.json
|
healed/grid_math/reap_keep75_s1226.console.log
ADDED
|
@@ -0,0 +1,284 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: setting up run wo1fbid9
|
| 6 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 7 |
+
wandb: Run data is saved locally in outputs/healed/grid_math/reap_keep75_s1226/wandb/run-20260716_194622-wo1fbid9
|
| 8 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 9 |
+
wandb: Syncing run reap-math-keep75-s1226
|
| 10 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 11 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/wo1fbid9
|
| 12 |
+
|
| 13 |
+
12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 5.31B | teacher overlap=False
|
| 14 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.09029728901929532, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 2.078125, "lr": 6e-06, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.9, "frames": {"chat": 254}, "mem_gb": 21.82}
|
| 15 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 16 |
+
[eval step 1] sample: 'To solve this problem, we need to understand the geometric implications of connecting the midpoints of the sides of a triangle. This process forms a new triangle called the Varicafo triangle, which is'
|
| 17 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.10507463545134912, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 1.8515625, "lr": 9e-06, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 241}, "mem_gb": 22.07}
|
| 18 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.10182668693264325, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 1.671875, "lr": 1.2e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.3, "frames": {"chat": 213}, "mem_gb": 22.1}
|
| 19 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08138736556212728, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 1.234375, "lr": 1.5e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 221}, "mem_gb": 22.15}
|
| 20 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08653134391192968, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 1.125, "lr": 1.8e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 196}, "mem_gb": 22.11}
|
| 21 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05705227090573559, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 0.87109375, "lr": 2.1e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.5, "frames": {"chat": 270}, "mem_gb": 21.91}
|
| 22 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0642389781346855, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 0.79296875, "lr": 2.4e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 216}, "mem_gb": 22.09}
|
| 23 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.056490197068297617, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 0.57421875, "lr": 2.7000000000000002e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 200}, "mem_gb": 22.06}
|
| 24 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05389531436835726, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 0.5859375, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 206}, "mem_gb": 22.01}
|
| 25 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04393539156511737, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 234}, "mem_gb": 22.04}
|
| 26 |
+
[eval step 10] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and how the midpoints of its sides relate to the perimeter.\n\n### Steps to Solve:\n\n1. **Understand the Problem:**\n '
|
| 27 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04699371950745893, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.9, "frames": {"chat": 215}, "mem_gb": 22.05}
|
| 28 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0387724893038782, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 255}, "mem_gb": 22.04}
|
| 29 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.039335763697124396, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.4609375, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.5, "frames": {"chat": 250}, "mem_gb": 21.92}
|
| 30 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04199762986401717, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.447265625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 242}, "mem_gb": 22.09}
|
| 31 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0501405522093953, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.48046875, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 199}, "mem_gb": 22.1}
|
| 32 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.10351084683185133, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.8359375, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.9, "frames": {"chat": 208}, "mem_gb": 22.13}
|
| 33 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05745811533636103, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 208}, "mem_gb": 22.07}
|
| 34 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05604706888574486, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 209}, "mem_gb": 22.22}
|
| 35 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03987406260093364, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.1, "frames": {"chat": 235}, "mem_gb": 22.05}
|
| 36 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04426209665209365, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 204}, "mem_gb": 22.04}
|
| 37 |
+
[eval step 20] sample: 'To solve this problem, we need to understand the geometric transformation that occurs when the midpoints of the sides of a triangle are connected by segments. This transformation is known as the Varig'
|
| 38 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05017195799251398, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 208}, "mem_gb": 22.1}
|
| 39 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04144187349904484, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.2, "frames": {"chat": 240}, "mem_gb": 22.1}
|
| 40 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03879994131438434, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 246}, "mem_gb": 22.09}
|
| 41 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04156417465102859, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.361328125, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 243}, "mem_gb": 21.91}
|
| 42 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04885457618216363, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.7, "frames": {"chat": 208}, "mem_gb": 22.11}
|
| 43 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05428559299677921, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 219}, "mem_gb": 22.1}
|
| 44 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04315272529536548, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.5, "frames": {"chat": 211}, "mem_gb": 22.11}
|
| 45 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.047420875581144356, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.3, "frames": {"chat": 232}, "mem_gb": 22.07}
|
| 46 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.055055282410513606, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 214}, "mem_gb": 22.1}
|
| 47 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.038301955478964374, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.6, "frames": {"chat": 226}, "mem_gb": 21.99}
|
| 48 |
+
[eval step 30] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected by segments. This process creates a new triangle, known as th'
|
| 49 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03517087419158779, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.8, "frames": {"chat": 210}, "mem_gb": 22.11}
|
| 50 |
+
{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.038657001748944944, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 218}, "mem_gb": 21.93}
|
| 51 |
+
{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03476824780826767, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.2, "frames": {"chat": 233}, "mem_gb": 22.08}
|
| 52 |
+
{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.04166795010352507, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.333984375, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.6, "frames": {"chat": 215}, "mem_gb": 22.1}
|
| 53 |
+
{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03495994614987479, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.1, "frames": {"chat": 233}, "mem_gb": 22.09}
|
| 54 |
+
{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03733370227565368, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 209}, "mem_gb": 22.04}
|
| 55 |
+
{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02926472805084971, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.0, "frames": {"chat": 262}, "mem_gb": 21.97}
|
| 56 |
+
{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.031572706325010705, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.7, "frames": {"chat": 249}, "mem_gb": 22.06}
|
| 57 |
+
{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0411027745464351, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.6, "frames": {"chat": 227}, "mem_gb": 22.09}
|
| 58 |
+
{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.031158137715724298, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.1, "frames": {"chat": 221}, "mem_gb": 22.09}
|
| 59 |
+
[eval step 40] sample: 'To solve this problem, we need to understand the geometric transformation that occurs when the midpoints of the sides of a triangle are connected by segments. This transformation is known as the Varig'
|
| 60 |
+
{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.030553325068981698, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.298828125, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.3, "frames": {"chat": 234}, "mem_gb": 22.11}
|
| 61 |
+
{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.033070274770678955, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 213}, "mem_gb": 22.05}
|
| 62 |
+
{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02980018953961941, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 213}, "mem_gb": 21.99}
|
| 63 |
+
{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.026309455881919713, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.8, "frames": {"chat": 234}, "mem_gb": 22.02}
|
| 64 |
+
{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02914307115934013, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 222}, "mem_gb": 22.09}
|
| 65 |
+
{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.032646984772880874, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.298828125, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.1, "frames": {"chat": 227}, "mem_gb": 22.1}
|
| 66 |
+
{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03420899780319693, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.3046875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.7, "frames": {"chat": 218}, "mem_gb": 22.14}
|
| 67 |
+
{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03190279059541257, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.3, "frames": {"chat": 223}, "mem_gb": 22.11}
|
| 68 |
+
{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03304023641222156, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 206}, "mem_gb": 22.1}
|
| 69 |
+
{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.027658292231538022, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 213}, "mem_gb": 22.02}
|
| 70 |
+
[eval step 50] sample: 'To solve this problem, we need to understand the geometric transformation that occurs when the midpoints of the sides of a triangle are connected by segments. This transformation is known as the Varig'
|
| 71 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep75_s1226/step0050
|
| 72 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03513334818736961, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.2, "frames": {"chat": 223}, "mem_gb": 21.96}
|
| 73 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.029408954116201494, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 227}, "mem_gb": 22.07}
|
| 74 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.02545005696103132, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 253}, "mem_gb": 22.1}
|
| 75 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.025285162550009167, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.2353515625, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 216}, "mem_gb": 22.1}
|
| 76 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.020139218003411466, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.2236328125, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 205}, "mem_gb": 22.07}
|
| 77 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.023881658379912067, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.224609375, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 207}, "mem_gb": 22.16}
|
| 78 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02430247613317333, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 215}, "mem_gb": 22.08}
|
| 79 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019716258118348196, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.23046875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 228}, "mem_gb": 22.1}
|
| 80 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.022690464227786287, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 221}, "mem_gb": 22.14}
|
| 81 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01674839036881458, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.208984375, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.6, "frames": {"chat": 254}, "mem_gb": 21.93}
|
| 82 |
+
[eval step 60] sample: 'To solve this problem, we need to understand the geometric transformation that occurs when the midpoints of the sides of a triangle are connected by segments. This transformation is known as the Varig'
|
| 83 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02067672042537791, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.248046875, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 210}, "mem_gb": 22.06}
|
| 84 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01884686708585359, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 226}, "mem_gb": 22.02}
|
| 85 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.024455731499303753, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.236328125, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 212}, "mem_gb": 22.09}
|
| 86 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.022958892802521585, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.2275390625, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 211}, "mem_gb": 22.02}
|
| 87 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.023287332984805106, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 196}, "mem_gb": 22.07}
|
| 88 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02202161232800378, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.228515625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.1, "frames": {"chat": 212}, "mem_gb": 22.09}
|
| 89 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019485318966264216, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.2, "frames": {"chat": 244}, "mem_gb": 22.0}
|
| 90 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019222266574925744, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.1923828125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.1, "frames": {"chat": 222}, "mem_gb": 22.05}
|
| 91 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017886470882171608, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 218}, "mem_gb": 22.1}
|
| 92 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02026407581908473, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.21484375, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.2, "frames": {"chat": 252}, "mem_gb": 21.97}
|
| 93 |
+
[eval step 70] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected by segments. This construction forms a new triangle, known as'
|
| 94 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.024798149222143306, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.2490234375, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 202}, "mem_gb": 22.15}
|
| 95 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02727948643124352, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.2734375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.8, "frames": {"chat": 237}, "mem_gb": 22.1}
|
| 96 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02127351866173946, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.2080078125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 234}, "mem_gb": 22.08}
|
| 97 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018867127263825386, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.2294921875, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 215}, "mem_gb": 22.1}
|
| 98 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01688704885452753, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.19921875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 234}, "mem_gb": 22.03}
|
| 99 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018125394146431547, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.2060546875, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 216}, "mem_gb": 22.08}
|
| 100 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019315797305929784, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.2109375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 210}, "mem_gb": 22.05}
|
| 101 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018800026949353437, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.19921875, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 199}, "mem_gb": 22.1}
|
| 102 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017173049212961148, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.1943359375, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 210}, "mem_gb": 22.11}
|
| 103 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01991477519835656, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 225}, "mem_gb": 22.05}
|
| 104 |
+
[eval step 80] sample: 'To solve this problem, we need to understand the geometric transformation that occurs when the midpoints of the sides of a triangle are connected by segments. This transformation is known as the Varig'
|
| 105 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01865955205159262, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.2021484375, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.7, "frames": {"chat": 253}, "mem_gb": 21.95}
|
| 106 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017519566944839124, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 247}, "mem_gb": 22.07}
|
| 107 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01791032001193768, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.1875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 238}, "mem_gb": 22.07}
|
| 108 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01933542483012037, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 235}, "mem_gb": 22.09}
|
| 109 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 110 |
+
warnings.warn('Grouped GEMM not available.')
|
| 111 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 112 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 113 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 114 |
+
wandb: Run data is saved locally in outputs/healed/grid_math/reap_keep75_s1226/wandb/run-20260716_210030-s3v8iby3
|
| 115 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 116 |
+
wandb: Syncing run reap-math-keep75-s1226
|
| 117 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 118 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/s3v8iby3
|
| 119 |
+
|
| 120 |
+
resumed student weights from outputs/healed/grid_math/reap_keep75_s1226/step0050 (fresh optimizer, step counter at 0)
|
| 121 |
+
12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 5.31B | teacher overlap=False
|
| 122 |
+
restored optimizer/scheduler state from step 50; rebuilt 228 paged buffers
|
| 123 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.03513334818736961, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 64.2, "frames": {"chat": 223}, "mem_gb": 21.8}
|
| 124 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.029426557167852297, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.7, "frames": {"chat": 227}, "mem_gb": 22.07}
|
| 125 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.025465690592854903, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 253}, "mem_gb": 22.1}
|
| 126 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.025248298151674682, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.2353515625, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 216}, "mem_gb": 22.1}
|
| 127 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02003598742996498, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 205}, "mem_gb": 22.07}
|
| 128 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.023943437644823765, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.2255859375, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 207}, "mem_gb": 22.16}
|
| 129 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.024196449637723466, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 215}, "mem_gb": 22.08}
|
| 130 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01976318524889648, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 228}, "mem_gb": 22.1}
|
| 131 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.022648578603340623, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.244140625, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 221}, "mem_gb": 22.14}
|
| 132 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016758610638797595, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.0, "frames": {"chat": 254}, "mem_gb": 21.94}
|
| 133 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 134 |
+
[eval step 60] sample: 'To solve this problem, we need to understand the geometric transformation that occurs when the midpoints of the sides of a triangle are connected by segments. This transformation is known as the Varig'
|
| 135 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.020692213217533815, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 210}, "mem_gb": 22.06}
|
| 136 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0188163351556131, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.197265625, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.2, "frames": {"chat": 226}, "mem_gb": 22.02}
|
| 137 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.024447075344540645, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.2353515625, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.1, "frames": {"chat": 212}, "mem_gb": 22.09}
|
| 138 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02292411018896382, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.9, "frames": {"chat": 211}, "mem_gb": 22.02}
|
| 139 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.023229272329698628, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 196}, "mem_gb": 22.07}
|
| 140 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.021988844963729694, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.23046875, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.5, "frames": {"chat": 212}, "mem_gb": 22.09}
|
| 141 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01947804330990184, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.2119140625, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 244}, "mem_gb": 22.0}
|
| 142 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019210793147748335, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 222}, "mem_gb": 22.05}
|
| 143 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017839086343340266, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.9, "frames": {"chat": 218}, "mem_gb": 22.1}
|
| 144 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.020209834577407066, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.216796875, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.5, "frames": {"chat": 252}, "mem_gb": 21.97}
|
| 145 |
+
[eval step 70] sample: 'To solve this problem, we need to understand the geometric properties involved. When the midpoints of the sides of a triangle are connected, the resulting figure is known as the Varignon graph of the '
|
| 146 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02483861458793593, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.2490234375, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 202}, "mem_gb": 22.15}
|
| 147 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.027298253755876793, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 237}, "mem_gb": 22.1}
|
| 148 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.021423529687159073, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.2109375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.9, "frames": {"chat": 234}, "mem_gb": 22.08}
|
| 149 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01893475244011109, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.2373046875, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.2, "frames": {"chat": 215}, "mem_gb": 22.1}
|
| 150 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016884281579998786, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.1982421875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 234}, "mem_gb": 22.03}
|
| 151 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018117627698734092, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.2080078125, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 216}, "mem_gb": 22.08}
|
| 152 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019387165744726858, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.3, "frames": {"chat": 210}, "mem_gb": 22.05}
|
| 153 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018806892693745126, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.2001953125, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.4, "frames": {"chat": 199}, "mem_gb": 22.1}
|
| 154 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017233077865753635, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.203125, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 210}, "mem_gb": 22.11}
|
| 155 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019973451530816966, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.9, "frames": {"chat": 225}, "mem_gb": 22.05}
|
| 156 |
+
[eval step 80] sample: 'To solve this problem, we need to understand the geometric transformation that occurs when the midpoints of the sides of a triangle are connected by segments. This transformation is known as the Varig'
|
| 157 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01866991977209691, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.2021484375, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 253}, "mem_gb": 21.95}
|
| 158 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017613753451538892, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.4, "frames": {"chat": 247}, "mem_gb": 22.07}
|
| 159 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.017965144219622017, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.1884765625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 238}, "mem_gb": 22.07}
|
| 160 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01940085504942108, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 235}, "mem_gb": 22.09}
|
| 161 |
+
{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.020727789660030974, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.228515625, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 215}, "mem_gb": 22.06}
|
| 162 |
+
{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.020069647663552313, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.4, "frames": {"chat": 268}, "mem_gb": 22.07}
|
| 163 |
+
{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018086951273384815, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 228}, "mem_gb": 22.1}
|
| 164 |
+
{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01940715615034569, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.1, "frames": {"chat": 252}, "mem_gb": 22.03}
|
| 165 |
+
{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.020324613629250476, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.3, "frames": {"chat": 223}, "mem_gb": 22.11}
|
| 166 |
+
{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.024420837492410405, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.2236328125, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.8, "frames": {"chat": 226}, "mem_gb": 22.09}
|
| 167 |
+
[eval step 90] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected by segments. This configuration forms a new triangle, known a'
|
| 168 |
+
{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.022695899091350537, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.7, "frames": {"chat": 208}, "mem_gb": 22.14}
|
| 169 |
+
{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01997075284306581, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.2392578125, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.0, "frames": {"chat": 240}, "mem_gb": 22.03}
|
| 170 |
+
{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0180018085974191, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 222}, "mem_gb": 22.02}
|
| 171 |
+
{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015846047878447764, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.19140625, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 236}, "mem_gb": 22.09}
|
| 172 |
+
{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018581686680119794, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.2236328125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 217}, "mem_gb": 22.06}
|
| 173 |
+
{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.018064459037513008, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.236328125, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 252}, "mem_gb": 21.97}
|
| 174 |
+
{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016805086109414696, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 222}, "mem_gb": 22.09}
|
| 175 |
+
{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016599261284115104, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.901, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.9, "frames": {"chat": 242}, "mem_gb": 21.97}
|
| 176 |
+
{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.022208455129068654, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.7, "frames": {"chat": 219}, "mem_gb": 22.03}
|
| 177 |
+
{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.015049857790356813, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.177734375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 223}, "mem_gb": 22.04}
|
| 178 |
+
[eval step 100] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected. This construction results in a new triangle called the Varig'
|
| 179 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep75_s1226/step0100
|
| 180 |
+
{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.02226159777260618, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.19921875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 232}, "mem_gb": 22.04}
|
| 181 |
+
{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.020989066664020842, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.2421875, "lr": 3e-05, "finish_rate": 0.832, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 220}, "mem_gb": 22.1}
|
| 182 |
+
{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.019145282805276415, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.1884765625, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 210}, "mem_gb": 22.14}
|
| 183 |
+
{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01685496985814146, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.1826171875, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 226}, "mem_gb": 22.07}
|
| 184 |
+
{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.01805352083449252, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.203125, "lr": 3e-05, "finish_rate": 0.741, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.0, "frames": {"chat": 212}, "mem_gb": 22.09}
|
| 185 |
+
{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.016914581290911883, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.201171875, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.8, "frames": {"chat": 236}, "mem_gb": 22.11}
|
| 186 |
+
{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013538020706743313, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.1953125, "lr": 3e-05, "finish_rate": 0.928, "comp_len": 454.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.0, "frames": {"chat": 264}, "mem_gb": 21.98}
|
| 187 |
+
{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02562480551255091, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.4, "frames": {"chat": 229}, "mem_gb": 22.08}
|
| 188 |
+
{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012043519912695047, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.16015625, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 465.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.2, "frames": {"chat": 258}, "mem_gb": 21.95}
|
| 189 |
+
{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.017032683950662614, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.3, "frames": {"chat": 208}, "mem_gb": 22.11}
|
| 190 |
+
[eval step 110] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected. This construction results in a new triangle called the Varig'
|
| 191 |
+
{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012560956759933227, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.150390625, "lr": 3e-05, "finish_rate": 0.88, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.2, "frames": {"chat": 249}, "mem_gb": 22.02}
|
| 192 |
+
{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012023984959484854, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.1474609375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 220}, "mem_gb": 22.09}
|
| 193 |
+
{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01255449386528538, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.1630859375, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.4, "frames": {"chat": 223}, "mem_gb": 22.08}
|
| 194 |
+
{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014240974927445253, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.171875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.9, "frames": {"chat": 221}, "mem_gb": 22.09}
|
| 195 |
+
{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011515381241531578, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.1416015625, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.6, "frames": {"chat": 230}, "mem_gb": 22.0}
|
| 196 |
+
{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01467221461532948, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.1552734375, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.4, "frames": {"chat": 214}, "mem_gb": 22.07}
|
| 197 |
+
{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.017758685411163606, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.1982421875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.8, "frames": {"chat": 214}, "mem_gb": 22.09}
|
| 198 |
+
{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015415362640703097, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.1669921875, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 210}, "mem_gb": 22.13}
|
| 199 |
+
{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.016011621292408865, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.177734375, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 214}, "mem_gb": 22.09}
|
| 200 |
+
{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01363702624545355, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.1591796875, "lr": 3e-05, "finish_rate": 0.791, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 215}, "mem_gb": 22.05}
|
| 201 |
+
[eval step 120] sample: 'To solve this problem, we need to understand the geometric properties involved. When the midpoints of the sides of a triangle are connected, these segments form a new triangle called the Varignon para'
|
| 202 |
+
{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015719523386086804, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.1943359375, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 208}, "mem_gb": 22.09}
|
| 203 |
+
{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014719627378954707, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.1845703125, "lr": 3e-05, "finish_rate": 0.789, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 218}, "mem_gb": 21.97}
|
| 204 |
+
{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011802848596102557, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.138671875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 233}, "mem_gb": 21.99}
|
| 205 |
+
{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011568954351439606, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.142578125, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 231}, "mem_gb": 22.03}
|
| 206 |
+
{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013774385743390303, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.162109375, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 235}, "mem_gb": 22.23}
|
| 207 |
+
{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01195954483349342, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.15625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.3, "frames": {"chat": 216}, "mem_gb": 22.08}
|
| 208 |
+
{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01140332082757183, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.15234375, "lr": 3e-05, "finish_rate": 0.831, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.4, "frames": {"chat": 237}, "mem_gb": 22.11}
|
| 209 |
+
{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015267859172041062, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.1728515625, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 203}, "mem_gb": 22.11}
|
| 210 |
+
{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011874694500351324, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.1513671875, "lr": 3e-05, "finish_rate": 0.873, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 236}, "mem_gb": 22.13}
|
| 211 |
+
{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01162539423473257, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.1640625, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.8, "frames": {"chat": 214}, "mem_gb": 22.1}
|
| 212 |
+
[eval step 130] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected by segments. This configuration forms a new triangle, known a'
|
| 213 |
+
{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014787918717813833, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.181640625, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 209}, "mem_gb": 22.09}
|
| 214 |
+
{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013407840112701524, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.1650390625, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.9, "frames": {"chat": 256}, "mem_gb": 22.1}
|
| 215 |
+
{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011451891991806526, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.1455078125, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.0, "frames": {"chat": 230}, "mem_gb": 21.96}
|
| 216 |
+
{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013753886991828525, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.1796875, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 230}, "mem_gb": 22.15}
|
| 217 |
+
{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01416524691574353, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.1630859375, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.2, "frames": {"chat": 227}, "mem_gb": 22.05}
|
| 218 |
+
{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.015579758063218712, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.1591796875, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 208}, "mem_gb": 22.11}
|
| 219 |
+
{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01355467732361673, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.1416015625, "lr": 3e-05, "finish_rate": 0.699, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 206}, "mem_gb": 22.13}
|
| 220 |
+
{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013072437331503413, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.1533203125, "lr": 3e-05, "finish_rate": 0.82, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 228}, "mem_gb": 22.0}
|
| 221 |
+
{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014515860647708177, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.1474609375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.7, "frames": {"chat": 224}, "mem_gb": 22.09}
|
| 222 |
+
{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.014361987802867467, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.171875, "lr": 3e-05, "finish_rate": 0.66, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 200}, "mem_gb": 22.13}
|
| 223 |
+
[eval step 140] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected by segments. This configuration forms a new triangle, known a'
|
| 224 |
+
{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013031911173191232, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.15625, "lr": 3e-05, "finish_rate": 0.714, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 196}, "mem_gb": 22.11}
|
| 225 |
+
{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012162300161850484, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.142578125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.2, "frames": {"chat": 223}, "mem_gb": 22.09}
|
| 226 |
+
{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.013919239423350277, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.169921875, "lr": 3e-05, "finish_rate": 0.869, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 213}, "mem_gb": 21.98}
|
| 227 |
+
{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0119247377780984, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.138671875, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 232}, "mem_gb": 22.03}
|
| 228 |
+
{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01081809983511921, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.1376953125, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 223}, "mem_gb": 22.02}
|
| 229 |
+
{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.011735041455866303, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.1328125, "lr": 3e-05, "finish_rate": 0.85, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.1, "frames": {"chat": 233}, "mem_gb": 22.12}
|
| 230 |
+
{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01394764332479875, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.1630859375, "lr": 3e-05, "finish_rate": 0.816, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 217}, "mem_gb": 22.11}
|
| 231 |
+
{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.017427034268093607, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.171875, "lr": 3e-05, "finish_rate": 0.752, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.3, "frames": {"chat": 202}, "mem_gb": 22.17}
|
| 232 |
+
{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.01164601769730992, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.1376953125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.7, "frames": {"chat": 253}, "mem_gb": 22.03}
|
| 233 |
+
{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.012088888911770967, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.150390625, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.8, "frames": {"chat": 231}, "mem_gb": 22.03}
|
| 234 |
+
[eval step 150] sample: 'To solve this problem, we need to understand the geometric properties involved when the midpoints of the sides of a triangle are connected by segments. This configuration forms a new triangle, known a'
|
| 235 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/reap_keep75_s1226/step0150
|
| 236 |
+
wandb: updating run metadata
|
| 237 |
+
wandb: uploading wandb-summary.json; uploading config.yaml; uploading output.log
|
| 238 |
+
wandb:
|
| 239 |
+
wandb: Run history:
|
| 240 |
+
wandb: comp_len ▆▄▁▄▆▆▅▅▂▇▅▅▆█▃▂▄▅▂▅▄▄▁▂▄▆▆▆▅▃▅▃▃▆▁▆▇▄█▃
|
| 241 |
+
wandb: cumulative_loss_tokens ▁▁▁▂▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▄▅▅▅▅▅▅▆▆▆▇▇▇▇▇████
|
| 242 |
+
wandb: epoch ▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅███████████████
|
| 243 |
+
wandb: finish_rate ▇▄▅▃▇▃▄█▆▅█▇▆▆█▅▆█▅▇▃▆█▆▆▄▄▃▅▃▇▅▂▁▂▇▆▆▅▇
|
| 244 |
+
wandb: forward_topk_kl █▆▅▄▅▃▄▃▃▃▃▃▃▃▄▃▄▅▄▄▂▃▃▃▃▅▃▁▁▃▁▂▁▂▂▂▂▁▂▁
|
| 245 |
+
wandb: grad_norm ▄▄█▅▄▃▄▅▃▄▃▅▃▂▃▅▃▄▄▃▇▄▃▂▃▂▁▂▁▂▁▂▁▂▁▂▁▂▁▁
|
| 246 |
+
wandb: lr ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 247 |
+
wandb: mem_gb ▅▅▄▆▄▄▄▂▃▃▃▁▄▄▄▅▅▆▄▁▃▆▄▅▅▄▃▂▅▃▁▂█▅▅▅▅▅▃▃
|
| 248 |
+
wandb: step ▁▁▁▁▁▂▂▂▂▂▂▂▃▃▃▃▄▄▄▄▄▄▄▅▅▆▆▆▆▆▆▆▇▇▇▇▇▇██
|
| 249 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 250 |
+
wandb: +3 ...
|
| 251 |
+
wandb:
|
| 252 |
+
wandb: Run summary:
|
| 253 |
+
wandb: comp_len 519.5
|
| 254 |
+
wandb: cumulative_loss_tokens 18000000
|
| 255 |
+
wandb: epoch 2
|
| 256 |
+
wandb: finish_rate 0.879
|
| 257 |
+
wandb: forward_topk_kl 0.01209
|
| 258 |
+
wandb: grad_norm 0.15039
|
| 259 |
+
wandb: lr 3e-05
|
| 260 |
+
wandb: mem_gb 22.03
|
| 261 |
+
wandb: step 150
|
| 262 |
+
wandb: t_data_s 0
|
| 263 |
+
wandb: +4 ...
|
| 264 |
+
wandb:
|
| 265 |
+
wandb: 🚀 View run reap-math-keep75-s1226 at: https://wandb.ai/hbfreed/glean-grid/runs/s3v8iby3
|
| 266 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid
|
| 267 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 268 |
+
wandb: Find logs at: outputs/healed/grid_math/reap_keep75_s1226/wandb/run-20260716_210030-s3v8iby3/logs
|
| 269 |
+
{
|
| 270 |
+
"correct": 892,
|
| 271 |
+
"accuracy": 0.6762699014404853,
|
| 272 |
+
"finished": 1316,
|
| 273 |
+
"finish_rate": 0.9977255496588324,
|
| 274 |
+
"mean_completion_tokens": 109.70583775587566
|
| 275 |
+
}
|
| 276 |
+
saved item-level results -> outputs/evals/grid_math/reap_keep75_s1226_step100_chat.json
|
| 277 |
+
{
|
| 278 |
+
"correct": 888,
|
| 279 |
+
"accuracy": 0.6732373009855952,
|
| 280 |
+
"finished": 1313,
|
| 281 |
+
"finish_rate": 0.9954510993176648,
|
| 282 |
+
"mean_completion_tokens": 112.45185746777862
|
| 283 |
+
}
|
| 284 |
+
saved item-level results -> outputs/evals/grid_math/reap_keep75_s1226_step150_chat.json
|
healed/grid_math/uniform_keep50_s1226.console.log
ADDED
|
@@ -0,0 +1,232 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: setting up run 4hqt5dvm
|
| 6 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 7 |
+
wandb: Run data is saved locally in outputs/healed/grid_math/uniform_keep50_s1226/wandb/run-20260716_000817-4hqt5dvm
|
| 8 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 9 |
+
wandb: Syncing run uniform-math-keep50-s1226
|
| 10 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-grid
|
| 11 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-grid/runs/4hqt5dvm
|
| 12 |
+
|
| 13 |
+
12115 cached top-128 chat trajectories / 6,476,634 unique tokens | 53 steps/epoch | 150 total steps | student params 3.70B | teacher overlap=False
|
| 14 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4376790891032666, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 5.6875, "lr": 6e-06, "finish_rate": 0.902, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.6, "frames": {"chat": 254}, "mem_gb": 15.82}
|
| 15 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 16 |
+
[eval step 1] sample: 'To solve this problem, we need to break it down into manageable steps and use algebraic methods to find the perimeter of the resulting triangle.\n\n1. **Identify the sides of the triangle:**\n Let the '
|
| 17 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4641113032187025, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 5.21875, "lr": 9e-06, "finish_rate": 0.876, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 241}, "mem_gb": 16.03}
|
| 18 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4724495052379866, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 4.34375, "lr": 1.2e-05, "finish_rate": 0.746, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.9, "frames": {"chat": 213}, "mem_gb": 16.05}
|
| 19 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.37771775155427556, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 2.90625, "lr": 1.5e-05, "finish_rate": 0.864, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.8, "frames": {"chat": 221}, "mem_gb": 16.1}
|
| 20 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.38108605218355857, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 2.375, "lr": 1.8e-05, "finish_rate": 0.745, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.2, "frames": {"chat": 196}, "mem_gb": 16.06}
|
| 21 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27793007747692366, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 1.546875, "lr": 2.1e-05, "finish_rate": 0.926, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.2, "frames": {"chat": 270}, "mem_gb": 15.86}
|
| 22 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2918766071258734, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 1.5859375, "lr": 2.4e-05, "finish_rate": 0.815, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.4, "frames": {"chat": 216}, "mem_gb": 16.04}
|
| 23 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27928091689758, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 1.34375, "lr": 2.7000000000000002e-05, "finish_rate": 0.775, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.0, "frames": {"chat": 200}, "mem_gb": 16.01}
|
| 24 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.24007244749739767, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 1.0625, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.1, "frames": {"chat": 206}, "mem_gb": 15.96}
|
| 25 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20880229707335432, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 0.82421875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 234}, "mem_gb": 15.99}
|
| 26 |
+
[eval step 10] sample: "To solve this problem, we need to understand the properties of a triangle formed by connecting the midpoints of its sides.\n\n### Step-by-Step Solution:\n\n1. **Identify the Midpoints:**\n Let's denote t"
|
| 27 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23035434478012223, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 0.7890625, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.8, "frames": {"chat": 215}, "mem_gb": 16.01}
|
| 28 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19400933720568816, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.3, "frames": {"chat": 255}, "mem_gb": 15.99}
|
| 29 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19387649352867156, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 250}, "mem_gb": 15.87}
|
| 30 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19658990038111807, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.63671875, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 242}, "mem_gb": 16.04}
|
| 31 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22262518489745756, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.6640625, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.7, "frames": {"chat": 199}, "mem_gb": 16.05}
|
| 32 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23559244946800173, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.703125, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.5, "frames": {"chat": 208}, "mem_gb": 16.08}
|
| 33 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.19350703061049182, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.6875, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.8, "frames": {"chat": 208}, "mem_gb": 16.02}
|
| 34 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.20726347525846214, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.5, "frames": {"chat": 209}, "mem_gb": 16.17}
|
| 35 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16786122204847634, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.8, "frames": {"chat": 235}, "mem_gb": 16.01}
|
| 36 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16816826735908785, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.0, "frames": {"chat": 204}, "mem_gb": 16.0}
|
| 37 |
+
[eval step 20] sample: "To solve this problem, let's break it down into manageable steps:\n\n1. **Understand the Problem:**\n - The perimeter of the original triangle is 28.\n - The midpoints of the sides of the triangle are"
|
| 38 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21450991363972424, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.71875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.6, "frames": {"chat": 208}, "mem_gb": 16.05}
|
| 39 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1640882762360076, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.4921875, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.7, "frames": {"chat": 240}, "mem_gb": 16.05}
|
| 40 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15884890472330154, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.7, "frames": {"chat": 246}, "mem_gb": 16.04}
|
| 41 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16209638715144248, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.48828125, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.3, "frames": {"chat": 243}, "mem_gb": 15.86}
|
| 42 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.17776700306013227, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.50390625, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.2, "frames": {"chat": 208}, "mem_gb": 16.06}
|
| 43 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16285465159372736, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 219}, "mem_gb": 16.05}
|
| 44 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18971304966118185, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.55859375, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.1, "frames": {"chat": 211}, "mem_gb": 16.06}
|
| 45 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1514321407922233, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.1, "frames": {"chat": 232}, "mem_gb": 16.02}
|
| 46 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16563739912273984, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.47265625, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 214}, "mem_gb": 16.05}
|
| 47 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1536090032674993, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.4375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.7, "frames": {"chat": 226}, "mem_gb": 15.95}
|
| 48 |
+
[eval step 30] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of its sides.\n\n1. **Identify the Midpoints:**\n Let the sides of the triangle '
|
| 49 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.14960331625776987, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 210}, "mem_gb": 16.06}
|
| 50 |
+
{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13398413514227916, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 218}, "mem_gb": 15.88}
|
| 51 |
+
{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13549648077127835, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.6, "frames": {"chat": 233}, "mem_gb": 16.04}
|
| 52 |
+
{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16946489836232115, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.1, "frames": {"chat": 215}, "mem_gb": 16.05}
|
| 53 |
+
{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15514036291707306, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.42578125, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 233}, "mem_gb": 16.04}
|
| 54 |
+
{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13463629507801186, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.7, "frames": {"chat": 209}, "mem_gb": 15.99}
|
| 55 |
+
{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1336245067174236, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.2, "frames": {"chat": 262}, "mem_gb": 15.92}
|
| 56 |
+
{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13221798140201718, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.392578125, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 249}, "mem_gb": 16.01}
|
| 57 |
+
{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16325980938548842, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.44921875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.4, "frames": {"chat": 227}, "mem_gb": 16.05}
|
| 58 |
+
{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1249977902659215, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 221}, "mem_gb": 16.04}
|
| 59 |
+
[eval step 40] sample: "To solve this problem, we need to understand the properties of the triangle formed by connecting the midpoints of its sides. Let's break down the problem step-by-step:\n\n1. **Identify the Midpoints:**\n"
|
| 60 |
+
{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13953979968397567, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.400390625, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 234}, "mem_gb": 16.06}
|
| 61 |
+
{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.11785674755799895, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.0, "frames": {"chat": 213}, "mem_gb": 16.0}
|
| 62 |
+
{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.11892015679478646, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.2, "frames": {"chat": 213}, "mem_gb": 15.94}
|
| 63 |
+
{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.12541690055585156, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.9, "frames": {"chat": 234}, "mem_gb": 15.97}
|
| 64 |
+
{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.12613885796476776, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 222}, "mem_gb": 16.04}
|
| 65 |
+
{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.14734216099033753, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.4296875, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.0, "frames": {"chat": 227}, "mem_gb": 16.06}
|
| 66 |
+
{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1316178731423492, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.1, "frames": {"chat": 218}, "mem_gb": 16.09}
|
| 67 |
+
{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1474023002519893, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.412109375, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 223}, "mem_gb": 16.06}
|
| 68 |
+
{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15133659735408922, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.423828125, "lr": 3e-05, "finish_rate": 0.772, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.7, "frames": {"chat": 206}, "mem_gb": 16.05}
|
| 69 |
+
{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13506432898584753, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.8, "frames": {"chat": 213}, "mem_gb": 15.97}
|
| 70 |
+
[eval step 50] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of its sides.\n\n### Steps to Solve:\n\n1. **Identify the Midpoints:**\n Let the s'
|
| 71 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep50_s1226/step0050
|
| 72 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16151001168893028, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 538.1, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 38.0, "frames": {"chat": 223}, "mem_gb": 15.91}
|
| 73 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1424709311401161, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.828, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 227}, "mem_gb": 16.02}
|
| 74 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.12403348642749092, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.40234375, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.9, "frames": {"chat": 253}, "mem_gb": 16.05}
|
| 75 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10975177370604748, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 216}, "mem_gb": 16.06}
|
| 76 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11024285306433837, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 585.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.4, "frames": {"chat": 205}, "mem_gb": 16.02}
|
| 77 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.1259988553530847, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 579.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 207}, "mem_gb": 16.11}
|
| 78 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.12079795586705828, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.375, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.8, "frames": {"chat": 215}, "mem_gb": 16.03}
|
| 79 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08849038569725429, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.7, "frames": {"chat": 228}, "mem_gb": 16.05}
|
| 80 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11468118678036457, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.747, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.3, "frames": {"chat": 221}, "mem_gb": 16.09}
|
| 81 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08303429898895945, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 254}, "mem_gb": 15.89}
|
| 82 |
+
[eval step 60] sample: "To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of its sides. Let's denote the sides of the original triangle as \\(a\\), \\(b\\), "
|
| 83 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08057470158524811, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.5, "frames": {"chat": 210}, "mem_gb": 16.01}
|
| 84 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08877109787901863, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 226}, "mem_gb": 15.97}
|
| 85 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09652406602411841, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 212}, "mem_gb": 16.04}
|
| 86 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10605174731106187, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.754, "comp_len": 568.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.4, "frames": {"chat": 211}, "mem_gb": 15.97}
|
| 87 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10443491208475704, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 33.9, "frames": {"chat": 196}, "mem_gb": 16.02}
|
| 88 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09040325381926571, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 212}, "mem_gb": 16.04}
|
| 89 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08616819082628936, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.8, "frames": {"chat": 244}, "mem_gb": 15.96}
|
| 90 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08744919102328519, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.3, "frames": {"chat": 222}, "mem_gb": 16.0}
|
| 91 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0974519415643687, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 218}, "mem_gb": 16.05}
|
| 92 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09689508310742677, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.3, "frames": {"chat": 252}, "mem_gb": 15.92}
|
| 93 |
+
[eval step 70] sample: "To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of its sides. Let's denote the sides of the triangle as \\(a\\), \\(b\\), and \\(c\\)"
|
| 94 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10549356051698948, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.34375, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.2, "frames": {"chat": 202}, "mem_gb": 16.1}
|
| 95 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11792290148728837, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.3, "frames": {"chat": 237}, "mem_gb": 16.05}
|
| 96 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09844771656909336, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.5, "frames": {"chat": 234}, "mem_gb": 16.03}
|
| 97 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07686399800172075, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.809, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 215}, "mem_gb": 16.05}
|
| 98 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0773060815339908, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 512.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 234}, "mem_gb": 15.98}
|
| 99 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07318647137211325, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.4, "frames": {"chat": 216}, "mem_gb": 16.03}
|
| 100 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08653065130996207, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.0, "frames": {"chat": 210}, "mem_gb": 16.0}
|
| 101 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08935221479797813, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.719, "comp_len": 603.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.9, "frames": {"chat": 199}, "mem_gb": 16.05}
|
| 102 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08349365643256654, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.3046875, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.7, "frames": {"chat": 210}, "mem_gb": 16.06}
|
| 103 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08468451300160959, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.33984375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 533.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 225}, "mem_gb": 16.0}
|
| 104 |
+
[eval step 80] sample: "To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of its sides. Let's denote the sides of the triangle as \\(a\\), \\(b\\), and \\(c\\)"
|
| 105 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08564345600182811, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.32421875, "lr": 3e-05, "finish_rate": 0.913, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.3, "frames": {"chat": 253}, "mem_gb": 15.9}
|
| 106 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08711425514323637, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.7, "frames": {"chat": 247}, "mem_gb": 16.02}
|
| 107 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08793842437754695, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.3125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.3, "frames": {"chat": 238}, "mem_gb": 16.02}
|
| 108 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09588153673959895, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 235}, "mem_gb": 16.04}
|
| 109 |
+
{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10150185753870755, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.3671875, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.5, "frames": {"chat": 215}, "mem_gb": 16.02}
|
| 110 |
+
{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09396262565804646, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.925, "comp_len": 447.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.5, "frames": {"chat": 268}, "mem_gb": 16.02}
|
| 111 |
+
{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0943427060281237, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.1, "frames": {"chat": 228}, "mem_gb": 16.05}
|
| 112 |
+
{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09217703740115588, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.3, "frames": {"chat": 252}, "mem_gb": 15.98}
|
| 113 |
+
{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08400688921958208, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.821, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.5, "frames": {"chat": 223}, "mem_gb": 16.06}
|
| 114 |
+
{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.11824734832737595, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.805, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.4, "frames": {"chat": 226}, "mem_gb": 16.05}
|
| 115 |
+
[eval step 90] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of its sides.\n\n1. **Identify the Midpoints:**\n Let the sides of the triangle '
|
| 116 |
+
{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10819188171550631, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.37109375, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.7, "frames": {"chat": 208}, "mem_gb": 16.1}
|
| 117 |
+
{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07604796628550005, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.294921875, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.6, "frames": {"chat": 240}, "mem_gb": 15.98}
|
| 118 |
+
{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.09279157262903949, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.842, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.7, "frames": {"chat": 222}, "mem_gb": 15.97}
|
| 119 |
+
{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08047183130540264, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 236}, "mem_gb": 16.05}
|
| 120 |
+
{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07721997782148732, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.296875, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.7, "frames": {"chat": 217}, "mem_gb": 16.01}
|
| 121 |
+
{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08794904841768245, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.365234375, "lr": 3e-05, "finish_rate": 0.921, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.9, "frames": {"chat": 252}, "mem_gb": 15.92}
|
| 122 |
+
{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.07570475751673802, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 540.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.0, "frames": {"chat": 222}, "mem_gb": 16.04}
|
| 123 |
+
{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08202633683364838, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.318359375, "lr": 3e-05, "finish_rate": 0.901, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.4, "frames": {"chat": 242}, "mem_gb": 15.92}
|
| 124 |
+
{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10986236275633177, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 547.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.9, "frames": {"chat": 219}, "mem_gb": 15.98}
|
| 125 |
+
{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08761272633089684, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.7, "frames": {"chat": 223}, "mem_gb": 15.99}
|
| 126 |
+
[eval step 100] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of its sides.\n\n1. **Identify the Triangle:**\n Let the triangle have sides \\(a'
|
| 127 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep50_s1226/step0100
|
| 128 |
+
{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08812294497648254, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.7, "frames": {"chat": 232}, "mem_gb": 15.99}
|
| 129 |
+
{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10124652686836198, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.359375, "lr": 3e-05, "finish_rate": 0.832, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.3, "frames": {"chat": 220}, "mem_gb": 16.05}
|
| 130 |
+
{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10257488710001732, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.7, "frames": {"chat": 210}, "mem_gb": 16.09}
|
| 131 |
+
{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0879882658485168, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.310546875, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.8, "frames": {"chat": 226}, "mem_gb": 16.02}
|
| 132 |
+
{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.08545675618657843, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.322265625, "lr": 3e-05, "finish_rate": 0.741, "comp_len": 566.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 212}, "mem_gb": 16.05}
|
| 133 |
+
{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0808614875221625, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.33203125, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.2, "frames": {"chat": 236}, "mem_gb": 16.06}
|
| 134 |
+
{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06216761510347327, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.31640625, "lr": 3e-05, "finish_rate": 0.928, "comp_len": 454.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 264}, "mem_gb": 15.93}
|
| 135 |
+
{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08583992769025887, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 229}, "mem_gb": 16.03}
|
| 136 |
+
{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06338514630446831, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 465.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.5, "frames": {"chat": 258}, "mem_gb": 15.9}
|
| 137 |
+
{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08646221952826405, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.314453125, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 208}, "mem_gb": 16.06}
|
| 138 |
+
[eval step 110] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of its sides.\n\n1. **Identify the Midpoints:**\n Let the sides of the triangle '
|
| 139 |
+
{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07044994304866996, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.88, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.7, "frames": {"chat": 249}, "mem_gb": 15.97}
|
| 140 |
+
{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.057520826121761154, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 545.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.3, "frames": {"chat": 220}, "mem_gb": 16.04}
|
| 141 |
+
{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07055744165470823, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.6, "frames": {"chat": 223}, "mem_gb": 16.03}
|
| 142 |
+
{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06006319779339246, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 221}, "mem_gb": 16.05}
|
| 143 |
+
{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.059337620800252386, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.2392578125, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 230}, "mem_gb": 15.95}
|
| 144 |
+
{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07029758505622546, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 214}, "mem_gb": 16.02}
|
| 145 |
+
{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08982678314192842, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.306640625, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.9, "frames": {"chat": 214}, "mem_gb": 16.04}
|
| 146 |
+
{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07058332749111577, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 571.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.5, "frames": {"chat": 210}, "mem_gb": 16.08}
|
| 147 |
+
{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07674362048539334, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.337890625, "lr": 3e-05, "finish_rate": 0.776, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.3, "frames": {"chat": 214}, "mem_gb": 16.05}
|
| 148 |
+
{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07251923356292149, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.26953125, "lr": 3e-05, "finish_rate": 0.791, "comp_len": 558.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 215}, "mem_gb": 16.0}
|
| 149 |
+
[eval step 120] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the effect of connecting the midpoints of its sides.\n\n1. **Understand the Problem:**\n - The perimeter of the'
|
| 150 |
+
{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08275286472147951, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.5, "frames": {"chat": 208}, "mem_gb": 16.04}
|
| 151 |
+
{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06121184289492667, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.2451171875, "lr": 3e-05, "finish_rate": 0.789, "comp_len": 550.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 218}, "mem_gb": 15.93}
|
| 152 |
+
{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05615919139813632, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.2392578125, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.7, "frames": {"chat": 233}, "mem_gb": 15.95}
|
| 153 |
+
{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.054271727860253305, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.2, "frames": {"chat": 231}, "mem_gb": 15.98}
|
| 154 |
+
{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06616332181881493, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 510.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.2, "frames": {"chat": 235}, "mem_gb": 16.18}
|
| 155 |
+
{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06655114342967669, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 555.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.7, "frames": {"chat": 216}, "mem_gb": 16.04}
|
| 156 |
+
{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.061825485060364005, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.30859375, "lr": 3e-05, "finish_rate": 0.831, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.3, "frames": {"chat": 237}, "mem_gb": 16.06}
|
| 157 |
+
{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07930022509256378, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.29296875, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 591.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.9, "frames": {"chat": 203}, "mem_gb": 16.06}
|
| 158 |
+
{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06389350613456529, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.873, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.7, "frames": {"chat": 236}, "mem_gb": 16.08}
|
| 159 |
+
{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06659580795313232, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.734, "comp_len": 560.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.8, "frames": {"chat": 214}, "mem_gb": 16.05}
|
| 160 |
+
[eval step 130] sample: 'To solve this problem, we need to understand the geometric properties of the triangle formed by connecting the midpoints of its sides.\n\n1. **Identify the Midpoints:**\n Let the triangle have sides \\('
|
| 161 |
+
{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06866884825111677, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.78, "comp_len": 574.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 209}, "mem_gb": 16.05}
|
| 162 |
+
{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.059468065135972574, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.0, "frames": {"chat": 256}, "mem_gb": 16.05}
|
| 163 |
+
{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06222362800457825, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.878, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.2, "frames": {"chat": 230}, "mem_gb": 15.91}
|
| 164 |
+
{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06654002138863628, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 521.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.5, "frames": {"chat": 230}, "mem_gb": 16.11}
|
| 165 |
+
{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07265007014510533, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.291015625, "lr": 3e-05, "finish_rate": 0.881, "comp_len": 528.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 227}, "mem_gb": 16.0}
|
| 166 |
+
{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06892339828287562, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.4, "frames": {"chat": 208}, "mem_gb": 16.06}
|
| 167 |
+
{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.07665983066726476, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.28515625, "lr": 3e-05, "finish_rate": 0.699, "comp_len": 582.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.8, "frames": {"chat": 206}, "mem_gb": 16.08}
|
| 168 |
+
{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06965464311223477, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.28125, "lr": 3e-05, "finish_rate": 0.82, "comp_len": 526.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.5, "frames": {"chat": 228}, "mem_gb": 15.95}
|
| 169 |
+
{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06695077130920253, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 535.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.5, "frames": {"chat": 224}, "mem_gb": 16.04}
|
| 170 |
+
{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06528048767484725, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.341796875, "lr": 3e-05, "finish_rate": 0.66, "comp_len": 600.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.9, "frames": {"chat": 200}, "mem_gb": 16.08}
|
| 171 |
+
[eval step 140] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the effect of connecting the midpoints of its sides.\n\n### Steps to Solve the Problem:\n\n1. **Understand the Geo'
|
| 172 |
+
{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.06874485386822683, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.27734375, "lr": 3e-05, "finish_rate": 0.714, "comp_len": 612.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 34.3, "frames": {"chat": 196}, "mem_gb": 16.06}
|
| 173 |
+
{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0601529593461814, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.2470703125, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 223}, "mem_gb": 16.04}
|
| 174 |
+
{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05905946633927524, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.26171875, "lr": 3e-05, "finish_rate": 0.869, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.1, "frames": {"chat": 213}, "mem_gb": 15.93}
|
| 175 |
+
{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05775715522489821, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.0, "frames": {"chat": 232}, "mem_gb": 15.98}
|
| 176 |
+
{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05793680476608376, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 223}, "mem_gb": 15.97}
|
| 177 |
+
{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.059874816510655605, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.30078125, "lr": 3e-05, "finish_rate": 0.85, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.9, "frames": {"chat": 233}, "mem_gb": 16.07}
|
| 178 |
+
{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0661496601765665, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.816, "comp_len": 553.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 35.8, "frames": {"chat": 217}, "mem_gb": 16.06}
|
| 179 |
+
{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.08860474907079091, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.752, "comp_len": 594.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.2, "frames": {"chat": 202}, "mem_gb": 16.12}
|
| 180 |
+
{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05917992416545749, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 38.4, "frames": {"chat": 253}, "mem_gb": 15.98}
|
| 181 |
+
{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.05659642454708616, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.259765625, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.6, "frames": {"chat": 231}, "mem_gb": 15.99}
|
| 182 |
+
[eval step 150] sample: 'To solve this problem, we need to understand the geometric properties of the triangle and the effect of connecting the midpoints of its sides.\n\n1. **Understand the Problem:**\n - The perimeter of the'
|
| 183 |
+
checkpoint snapshot queued -> outputs/healed/grid_math/uniform_keep50_s1226/step0150
|
| 184 |
+
wandb: updating run metadata
|
| 185 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 186 |
+
wandb:
|
| 187 |
+
wandb: Run history:
|
| 188 |
+
wandb: comp_len ▂▆▁▇█▇▇▆▄▆▅▆▅▇▇▃▅▆▂█▆▇█▄▅▇▅▆▄▁▅▅▄▆▆▂▇▅▆▄
|
| 189 |
+
wandb: cumulative_loss_tokens ▁▁▁▁▁▂▂▂▂▂▂▂▂▂▂▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▇▇▇▇▇▇█
|
| 190 |
+
wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅▅█████████████
|
| 191 |
+
wandb: finish_rate ▇▆▆█▄▃▇▇▆▄▆▃▅▃▇▆▃▇▁▅▃▇▄▄▁█▄▅▅█▄█▂▅▆▂▆▂▄▄
|
| 192 |
+
wandb: forward_topk_kl █▄▃▃▄▃▃▃▃▃▃▂▂▂▂▂▂▂▂▂▂▁▂▂▁▁▂▂▂▁▁▁▁▁▁▁▁▁▁▁
|
| 193 |
+
wandb: grad_norm █▇▄▂▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 194 |
+
wandb: lr ▁▃▆█████████████████████████████████████
|
| 195 |
+
wandb: mem_gb ▄▃▅█▅▅▃▅▃▃▅▃▅▄▅▃▂▃▅▃▄▅▄▅▄▃▅▂▅▂▆▅▁▅▆▅▆▃▅▃
|
| 196 |
+
wandb: step ▁▁▁▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▆▆▆▆▇▇▇▇▇█████
|
| 197 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁█▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 198 |
+
wandb: +3 ...
|
| 199 |
+
wandb:
|
| 200 |
+
wandb: Run summary:
|
| 201 |
+
wandb: comp_len 519.5
|
| 202 |
+
wandb: cumulative_loss_tokens 18000000
|
| 203 |
+
wandb: epoch 2
|
| 204 |
+
wandb: finish_rate 0.879
|
| 205 |
+
wandb: forward_topk_kl 0.0566
|
| 206 |
+
wandb: grad_norm 0.25977
|
| 207 |
+
wandb: lr 3e-05
|
| 208 |
+
wandb: mem_gb 15.99
|
| 209 |
+
wandb: step 150
|
| 210 |
+
wandb: t_data_s 0
|
| 211 |
+
wandb: +4 ...
|
| 212 |
+
wandb:
|
| 213 |
+
wandb: 🚀 View run uniform-math-keep50-s1226 at: https://wandb.ai/hbfreed/glean-grid/runs/4hqt5dvm
|
| 214 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-grid
|
| 215 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 216 |
+
wandb: Find logs at: outputs/healed/grid_math/uniform_keep50_s1226/wandb/run-20260716_000817-4hqt5dvm/logs
|
| 217 |
+
{
|
| 218 |
+
"correct": 672,
|
| 219 |
+
"accuracy": 0.5094768764215315,
|
| 220 |
+
"finished": 1306,
|
| 221 |
+
"finish_rate": 0.9901440485216073,
|
| 222 |
+
"mean_completion_tokens": 110.27141774071266
|
| 223 |
+
}
|
| 224 |
+
saved item-level results -> outputs/evals/grid_math/uniform_keep50_s1226_step100_chat.json
|
| 225 |
+
{
|
| 226 |
+
"correct": 666,
|
| 227 |
+
"accuracy": 0.5049279757391963,
|
| 228 |
+
"finished": 1311,
|
| 229 |
+
"finish_rate": 0.9939347990902199,
|
| 230 |
+
"mean_completion_tokens": 109.41622441243366
|
| 231 |
+
}
|
| 232 |
+
saved item-level results -> outputs/evals/grid_math/uniform_keep50_s1226_step150_chat.json
|
healed/grid_math/worker_s1224.log
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-16T00:08:10-07:00 [s1224] glean_keep50_s1224 already done, skip
|
| 2 |
+
2026-07-16T00:08:10-07:00 [s1224] healing uniform_keep50_s1224 on GPU-8ca70870
|
| 3 |
+
2026-07-16T01:54:22-07:00 [s1224] eval uniform_keep50_s1224 step100
|
| 4 |
+
2026-07-16T01:56:17-07:00 [s1224] eval uniform_keep50_s1224 step150
|
| 5 |
+
2026-07-16T01:58:03-07:00 [s1224] uniform_keep50_s1224 done -> 0.5178165276724791
|
| 6 |
+
2026-07-16T01:58:03-07:00 [s1224] healing reap_keep50_s1224 on GPU-8ca70870
|
| 7 |
+
2026-07-16T04:03:00-07:00 [s1224] eval reap_keep50_s1224 step100
|
| 8 |
+
2026-07-16T04:04:42-07:00 [s1224] eval reap_keep50_s1224 step150
|
| 9 |
+
2026-07-16T04:06:17-07:00 [s1224] reap_keep50_s1224 done -> 0.5890826383623957
|
| 10 |
+
2026-07-16T04:06:17-07:00 [s1224] healing glean_keep25_s1224 on GPU-8ca70870
|
| 11 |
+
2026-07-16T05:38:07-07:00 [s1224] eval glean_keep25_s1224 step100
|
| 12 |
+
2026-07-16T05:40:30-07:00 [s1224] eval glean_keep25_s1224 step150
|
| 13 |
+
2026-07-16T05:42:42-07:00 [s1224] glean_keep25_s1224 done -> 0.4313874147081122
|
| 14 |
+
2026-07-16T05:42:42-07:00 [s1224] healing uniform_keep25_s1224 on GPU-8ca70870
|
| 15 |
+
2026-07-16T07:06:54-07:00 [s1224] eval uniform_keep25_s1224 step100
|
| 16 |
+
2026-07-16T07:08:55-07:00 [s1224] eval uniform_keep25_s1224 step150
|
| 17 |
+
2026-07-16T07:10:56-07:00 [s1224] uniform_keep25_s1224 done -> 0.2100075815011372
|
| 18 |
+
2026-07-16T07:10:56-07:00 [s1224] healing reap_keep25_s1224 on GPU-8ca70870
|
| 19 |
+
2026-07-16T09:07:15-07:00 [s1224] eval reap_keep25_s1224 step100
|
| 20 |
+
2026-07-16T09:09:23-07:00 [s1224] eval reap_keep25_s1224 step150
|
| 21 |
+
2026-07-16T09:11:25-07:00 [s1224] reap_keep25_s1224 done -> 0.11827141774071266
|
| 22 |
+
2026-07-16T09:11:25-07:00 [s1224] healing glean_keep75_s1224 on GPU-8ca70870
|
healed/grid_math/worker_s1225.log
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-16T00:08:10-07:00 [s1225] glean_keep50_s1225 already done, skip
|
| 2 |
+
2026-07-16T00:08:10-07:00 [s1225] healing uniform_keep50_s1225 on GPU-a6acf07f
|
| 3 |
+
2026-07-16T01:42:46-07:00 [s1225] eval uniform_keep50_s1225 step100
|
| 4 |
+
2026-07-16T01:44:40-07:00 [s1225] eval uniform_keep50_s1225 step150
|
| 5 |
+
2026-07-16T01:46:28-07:00 [s1225] uniform_keep50_s1225 done -> 0.5125094768764216
|
| 6 |
+
2026-07-16T01:46:28-07:00 [s1225] healing reap_keep50_s1225 on GPU-a6acf07f
|
| 7 |
+
2026-07-16T03:45:15-07:00 [s1225] eval reap_keep50_s1225 step100
|
| 8 |
+
2026-07-16T03:46:58-07:00 [s1225] eval reap_keep50_s1225 step150
|
| 9 |
+
2026-07-16T03:48:37-07:00 [s1225] reap_keep50_s1225 done -> 0.5830174374526156
|
| 10 |
+
2026-07-16T03:48:37-07:00 [s1225] healing glean_keep25_s1225 on GPU-a6acf07f
|
| 11 |
+
2026-07-16T05:13:00-07:00 [s1225] eval glean_keep25_s1225 step100
|
| 12 |
+
2026-07-16T05:15:09-07:00 [s1225] eval glean_keep25_s1225 step150
|
| 13 |
+
2026-07-16T05:17:20-07:00 [s1225] glean_keep25_s1225 done -> 0.4275966641394996
|
| 14 |
+
2026-07-16T05:17:20-07:00 [s1225] healing uniform_keep25_s1225 on GPU-a6acf07f
|
| 15 |
+
2026-07-16T06:31:01-07:00 [s1225] eval uniform_keep25_s1225 step100
|
| 16 |
+
2026-07-16T06:33:00-07:00 [s1225] eval uniform_keep25_s1225 step150
|
| 17 |
+
2026-07-16T06:34:59-07:00 [s1225] uniform_keep25_s1225 done -> 0.22062168309325247
|
| 18 |
+
2026-07-16T06:34:59-07:00 [s1225] healing reap_keep25_s1225 on GPU-a6acf07f
|
| 19 |
+
2026-07-16T08:25:50-07:00 [s1225] eval reap_keep25_s1225 step100
|
| 20 |
+
2026-07-16T08:28:05-07:00 [s1225] eval reap_keep25_s1225 step150
|
| 21 |
+
2026-07-16T08:30:19-07:00 [s1225] reap_keep25_s1225 done -> 0.13495072024260804
|
| 22 |
+
2026-07-16T08:30:19-07:00 [s1225] healing glean_keep75_s1225 on GPU-a6acf07f
|
healed/keep50_offpolicy_warmup_s1224/args.json
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"student": "outputs/pruned/glean-0125inst-math-keep50",
|
| 3 |
+
"teacher": "allenai/OLMoE-1B-7B-0125-Instruct",
|
| 4 |
+
"training_mode": "off-policy",
|
| 5 |
+
"kl_direction": "forward",
|
| 6 |
+
"dataset": "allenai/RLVR-MATH",
|
| 7 |
+
"dataset_sources": null,
|
| 8 |
+
"max_difficulty": null,
|
| 9 |
+
"trajectories": "outputs/teacher_trajectories/dolci_math_curated.jsonl",
|
| 10 |
+
"trajectory_dataset": "allenai/Dolci-Instruct-RL",
|
| 11 |
+
"off_policy_frames": "chat",
|
| 12 |
+
"off_policy_max_seq_len": 2048,
|
| 13 |
+
"topk_targets": "outputs/teacher_trajectories/dolci_math_curated_opd_top128",
|
| 14 |
+
"max_loss_tokens": null,
|
| 15 |
+
"loss_tokens_per_step": 120000,
|
| 16 |
+
"teacher_device": "cuda:0",
|
| 17 |
+
"student_device": "cuda:0",
|
| 18 |
+
"lr": 3e-05,
|
| 19 |
+
"optimizer": "adamw8bit",
|
| 20 |
+
"weight_decay": 0.1,
|
| 21 |
+
"epochs": 3,
|
| 22 |
+
"prompts_per_step": 256,
|
| 23 |
+
"group_size": 1,
|
| 24 |
+
"rollout_batch": 64,
|
| 25 |
+
"micro_batch": 3,
|
| 26 |
+
"max_new_tokens": 256,
|
| 27 |
+
"max_prompt_len": 1024,
|
| 28 |
+
"warmup_steps": 10,
|
| 29 |
+
"max_grad_norm": 1.0,
|
| 30 |
+
"eval_every": 10,
|
| 31 |
+
"gsm8k_every": 0,
|
| 32 |
+
"gsm8k_n": 256,
|
| 33 |
+
"gsm8k_batch": 16,
|
| 34 |
+
"gsm8k_max_new_tokens": 512,
|
| 35 |
+
"gsm8k_frames": "chat",
|
| 36 |
+
"save_every": 50,
|
| 37 |
+
"out_dir": "outputs/healed/keep50_offpolicy_warmup_s1224",
|
| 38 |
+
"sweep": 150,
|
| 39 |
+
"wandb": true,
|
| 40 |
+
"wandb_project": "glean-heal",
|
| 41 |
+
"wandb_run_name": "offpolicy-warmup-keep50-s1224",
|
| 42 |
+
"wandb_run_id": null,
|
| 43 |
+
"wandb_resume": null,
|
| 44 |
+
"wandb_mode": "online",
|
| 45 |
+
"no_wandb_sync": false,
|
| 46 |
+
"debug": false,
|
| 47 |
+
"resume_from": null,
|
| 48 |
+
"start_step": 0,
|
| 49 |
+
"no_grad_checkpointing": false,
|
| 50 |
+
"seed": 1224,
|
| 51 |
+
"no_teacher_overlap": false,
|
| 52 |
+
"sync_checkpoints": false,
|
| 53 |
+
"rollout_engine": "hf",
|
| 54 |
+
"vllm_gpu": null,
|
| 55 |
+
"vllm_port": 8377,
|
| 56 |
+
"vllm_refresh_every": 5,
|
| 57 |
+
"vllm_serve_bin": "vllm-plugin/.venv/bin/python",
|
| 58 |
+
"vllm_gpu_mem_util": 0.85,
|
| 59 |
+
"liger_loss": false,
|
| 60 |
+
"gold_mix_lambda": 0.0,
|
| 61 |
+
"gold_topk_targets": null,
|
| 62 |
+
"gold_loss": "ce",
|
| 63 |
+
"gold_mix_decay": 0.0,
|
| 64 |
+
"fast_teacher": false,
|
| 65 |
+
"reference_kl_beta": 0.0,
|
| 66 |
+
"drop_truncated_rollouts": false,
|
| 67 |
+
"vllm_max_model_len": null,
|
| 68 |
+
"vllm_refresh_mode": "reload",
|
| 69 |
+
"vllm_live_dir": null,
|
| 70 |
+
"resolved_kl_direction": "forward"
|
| 71 |
+
}
|
healed/keep50_offpolicy_warmup_s1224/train_log.jsonl
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21866693885562322, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 0.703125, "lr": 6e-06, "finish_rate": 0.907, "comp_len": 508.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.337, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.4, "frames": {"chat": 236}, "mem_gb": 9.77, "mem_gb_teacher": 9.77}
|
| 2 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27999618121907116, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 0.80078125, "lr": 9e-06, "finish_rate": 0.781, "comp_len": 558.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.474, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.1, "frames": {"chat": 215}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 3 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3080657049433639, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 0.89453125, "lr": 1.2e-05, "finish_rate": 0.825, "comp_len": 553.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.376, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.3, "frames": {"chat": 217}, "mem_gb": 9.83, "mem_gb_teacher": 9.83}
|
| 4 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27968517109975216, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 0.80078125, "lr": 1.5e-05, "finish_rate": 0.8, "comp_len": 585.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.366, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.6, "frames": {"chat": 205}, "mem_gb": 9.89, "mem_gb_teacher": 9.89}
|
| 5 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2137425679458926, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 0.62890625, "lr": 1.8e-05, "finish_rate": 0.834, "comp_len": 524.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.318, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.8, "frames": {"chat": 229}, "mem_gb": 9.86, "mem_gb_teacher": 9.86}
|
| 6 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.34967619865822297, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 1.3125, "lr": 2.1e-05, "finish_rate": 0.812, "comp_len": 538.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.336, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 25.4, "frames": {"chat": 223}, "mem_gb": 9.93, "mem_gb_teacher": 9.93}
|
| 7 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23986013823635877, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 0.5546875, "lr": 2.4e-05, "finish_rate": 0.708, "comp_len": 594.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.42, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.6, "frames": {"chat": 202}, "mem_gb": 9.97, "mem_gb_teacher": 9.97}
|
| 8 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3001840955584, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 0.78515625, "lr": 2.7000000000000002e-05, "finish_rate": 0.77, "comp_len": 574.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.453, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 24.9, "frames": {"chat": 209}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 9 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21187542014177888, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 0.57421875, "lr": 3e-05, "finish_rate": 0.885, "comp_len": 528.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.397, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.7, "frames": {"chat": 227}, "mem_gb": 9.92, "mem_gb_teacher": 9.92}
|
| 10 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21670206268125525, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 0.6328125, "lr": 3e-05, "finish_rate": 0.848, "comp_len": 521.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.546, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.0, "frames": {"chat": 230}, "mem_gb": 9.99, "mem_gb_teacher": 9.99}
|
| 11 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21374244357372324, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 0.52734375, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.426, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 26.0, "frames": {"chat": 231}, "mem_gb": 9.84, "mem_gb_teacher": 9.84}
|
| 12 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.23642759951651096, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 0.625, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 489.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.246, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.3, "frames": {"chat": 245}, "mem_gb": 9.92, "mem_gb_teacher": 9.92}
|
| 13 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2684279385884603, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.73828125, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 571.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.375, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.7, "frames": {"chat": 210}, "mem_gb": 9.92, "mem_gb_teacher": 9.92}
|
| 14 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2720188724226008, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.69140625, "lr": 3e-05, "finish_rate": 0.758, "comp_len": 568.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.312, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.3, "frames": {"chat": 211}, "mem_gb": 9.92, "mem_gb_teacher": 9.92}
|
| 15 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2650659183566769, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.413, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.1, "frames": {"chat": 221}, "mem_gb": 9.98, "mem_gb_teacher": 9.98}
|
| 16 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22639379921114694, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.912, "comp_len": 480.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.366, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.1, "frames": {"chat": 250}, "mem_gb": 9.79, "mem_gb_teacher": 9.79}
|
| 17 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2573891951084137, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.796875, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 524.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.402, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 26.9, "frames": {"chat": 229}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 18 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22218937695200244, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.77734375, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 480.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.479, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.7, "frames": {"chat": 250}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 19 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.33289732446968556, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 2.546875, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 519.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.363, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.6, "frames": {"chat": 231}, "mem_gb": 9.81, "mem_gb_teacher": 9.81}
|
| 20 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.327336692000553, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 1.3984375, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 535.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.433, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 25.4, "frames": {"chat": 224}, "mem_gb": 9.86, "mem_gb_teacher": 9.86}
|
| 21 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.33612352709385257, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 1.640625, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.329, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.9, "frames": {"chat": 212}, "mem_gb": 9.9, "mem_gb_teacher": 9.9}
|
| 22 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2883151387684047, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 1.3984375, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 504.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.376, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.9, "frames": {"chat": 238}, "mem_gb": 9.85, "mem_gb_teacher": 9.85}
|
| 23 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.25279294178610046, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 1.3828125, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 466.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.445, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.9, "frames": {"chat": 257}, "mem_gb": 9.73, "mem_gb_teacher": 9.73}
|
| 24 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2615377183983723, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 1.5625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 528.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.405, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.7, "frames": {"chat": 227}, "mem_gb": 9.92, "mem_gb_teacher": 9.92}
|
| 25 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32016109869256615, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 2.546875, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 526.3, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.359, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.8, "frames": {"chat": 228}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 26 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.34886745701755084, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 2.0625, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 515.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.542, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.1, "frames": {"chat": 233}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 27 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3173991178593288, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 2.671875, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 515.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.351, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.3, "frames": {"chat": 233}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 28 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3661775833528489, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 2.984375, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 609.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.388, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.3, "frames": {"chat": 197}, "mem_gb": 10.03, "mem_gb_teacher": 10.03}
|
| 29 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.36429248579877116, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 5.625, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 502.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.384, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.0, "frames": {"chat": 239}, "mem_gb": 9.79, "mem_gb_teacher": 9.79}
|
| 30 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3706551998923222, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 4.34375, "lr": 3e-05, "finish_rate": 0.83, "comp_len": 535.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.302, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.1, "frames": {"chat": 224}, "mem_gb": 9.83, "mem_gb_teacher": 9.83}
|
| 31 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.35791126018886765, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 2.03125, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 553.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.388, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.3, "frames": {"chat": 217}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 32 |
+
{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3755841711225609, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 2.3125, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.259, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.2, "frames": {"chat": 241}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 33 |
+
{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3561015821622064, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 2.359375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.407, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.9, "frames": {"chat": 218}, "mem_gb": 9.92, "mem_gb_teacher": 9.92}
|
| 34 |
+
{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.38959850126380724, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 2.125, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.439, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.9, "frames": {"chat": 206}, "mem_gb": 9.93, "mem_gb_teacher": 9.93}
|
| 35 |
+
{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3970217802577963, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 2.734375, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 517.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.512, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.0, "frames": {"chat": 232}, "mem_gb": 9.97, "mem_gb_teacher": 9.97}
|
| 36 |
+
{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4204790102675557, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 1.796875, "lr": 3e-05, "finish_rate": 0.771, "comp_len": 550.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.432, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.6, "frames": {"chat": 218}, "mem_gb": 9.99, "mem_gb_teacher": 9.99}
|
| 37 |
+
{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.40933242611338694, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 1.6015625, "lr": 3e-05, "finish_rate": 0.779, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.428, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.9, "frames": {"chat": 213}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 38 |
+
{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3901712652951479, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 2.640625, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 483.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.527, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.5, "frames": {"chat": 248}, "mem_gb": 9.92, "mem_gb_teacher": 9.92}
|
| 39 |
+
{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3685633979354054, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 1.921875, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 550.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.333, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.2, "frames": {"chat": 218}, "mem_gb": 9.99, "mem_gb_teacher": 9.99}
|
| 40 |
+
{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.39544522463083265, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 2.3125, "lr": 3e-05, "finish_rate": 0.851, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.401, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.7, "frames": {"chat": 221}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 41 |
+
{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.38834569306795796, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 1.640625, "lr": 3e-05, "finish_rate": 0.894, "comp_len": 508.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.396, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.5, "frames": {"chat": 236}, "mem_gb": 9.87, "mem_gb_teacher": 9.87}
|
| 42 |
+
{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.40595541520963113, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 2.390625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 487.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.338, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.6, "frames": {"chat": 246}, "mem_gb": 9.79, "mem_gb_teacher": 9.79}
|
| 43 |
+
{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4013322958761205, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 2.703125, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 512.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.48, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.9, "frames": {"chat": 234}, "mem_gb": 10.05, "mem_gb_teacher": 10.05}
|
| 44 |
+
{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.41921593125065165, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 2.984375, "lr": 3e-05, "finish_rate": 0.748, "comp_len": 594.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.474, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.7, "frames": {"chat": 202}, "mem_gb": 9.93, "mem_gb_teacher": 9.93}
|
| 45 |
+
{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.45117179917966327, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 1.953125, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.36, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.2, "frames": {"chat": 217}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 46 |
+
{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.45246317840516564, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 4.0625, "lr": 3e-05, "finish_rate": 0.866, "comp_len": 535.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.47, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.4, "frames": {"chat": 224}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 47 |
+
{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.49074055876086153, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 2.265625, "lr": 3e-05, "finish_rate": 0.753, "comp_len": 558.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.285, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.2, "frames": {"chat": 215}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 48 |
+
{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.502812573158741, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 4.78125, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 463.3, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.356, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.6, "frames": {"chat": 259}, "mem_gb": 9.87, "mem_gb_teacher": 9.87}
|
| 49 |
+
{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4940076053115229, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 2.875, "lr": 3e-05, "finish_rate": 0.829, "comp_len": 571.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.475, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.5, "frames": {"chat": 210}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 50 |
+
{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.47275001460264127, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 3.296875, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.304, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.3, "frames": {"chat": 213}, "mem_gb": 9.99, "mem_gb_teacher": 9.99}
|
| 51 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4224179199380179, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 1.75, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 540.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.453, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.1, "frames": {"chat": 222}, "mem_gb": 9.9, "mem_gb_teacher": 9.9}
|
| 52 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4378351664955417, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 2.421875, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 510.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.439, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.7, "frames": {"chat": 235}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 53 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.44709739099716145, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 6.625, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.417, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.4, "frames": {"chat": 208}, "mem_gb": 9.91, "mem_gb_teacher": 9.91}
|
| 54 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4153756284924845, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 5.4375, "lr": 3e-05, "finish_rate": 0.733, "comp_len": 628.3, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.413, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 23.5, "frames": {"chat": 191}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 55 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.40591654521947107, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 5.75, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 547.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.523, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.1, "frames": {"chat": 219}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 56 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.43598461368133623, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 7.0, "lr": 3e-05, "finish_rate": 0.778, "comp_len": 579.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.459, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.8, "frames": {"chat": 207}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 57 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4643517450052003, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 5.53125, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.449, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.4, "frames": {"chat": 208}, "mem_gb": 9.91, "mem_gb_teacher": 9.91}
|
| 58 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.41939549018144606, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 26.125, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.288, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.7, "frames": {"chat": 219}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 59 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.44048424391622343, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 58.5, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 487.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.329, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.8, "frames": {"chat": 246}, "mem_gb": 9.82, "mem_gb_teacher": 9.82}
|
| 60 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4594272028216471, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 46.0, "lr": 3e-05, "finish_rate": 0.704, "comp_len": 582.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.437, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.3, "frames": {"chat": 206}, "mem_gb": 9.97, "mem_gb_teacher": 9.97}
|
| 61 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4421705829419196, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 5.1875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.375, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.2, "frames": {"chat": 233}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 62 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5114135434468587, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 20.375, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.363, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.4, "frames": {"chat": 229}, "mem_gb": 9.82, "mem_gb_teacher": 9.82}
|
| 63 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4925492258039614, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 22.0, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.285, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.7, "frames": {"chat": 236}, "mem_gb": 9.85, "mem_gb_teacher": 9.85}
|
| 64 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4923259206386904, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 21.875, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.37, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.0, "frames": {"chat": 239}, "mem_gb": 9.74, "mem_gb_teacher": 9.74}
|
| 65 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.41652837800706427, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 8.875, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.349, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.8, "frames": {"chat": 241}, "mem_gb": 9.86, "mem_gb_teacher": 9.86}
|
| 66 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3965861890381823, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 3.484375, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.44, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.1, "frames": {"chat": 226}, "mem_gb": 9.82, "mem_gb_teacher": 9.82}
|
| 67 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.36972922986969353, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 17.0, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.32, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.0, "frames": {"chat": 234}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 68 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.374694859992216, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 21.125, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.401, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.2, "frames": {"chat": 257}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 69 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4051088455612461, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 7.3125, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.523, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.1, "frames": {"chat": 208}, "mem_gb": 10.0, "mem_gb_teacher": 10.0}
|
| 70 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4079768711109956, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 2.046875, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.522, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.0, "frames": {"chat": 211}, "mem_gb": 9.97, "mem_gb_teacher": 9.97}
|
| 71 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3891906266813477, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 3.484375, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.407, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.1, "frames": {"chat": 227}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 72 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4033234915149709, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 5.03125, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.434, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.2, "frames": {"chat": 211}, "mem_gb": 9.93, "mem_gb_teacher": 9.93}
|
| 73 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3784339854914695, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 4.0, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.403, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.6, "frames": {"chat": 238}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 74 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.37151155275255443, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 1.3515625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.517, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.8, "frames": {"chat": 237}, "mem_gb": 9.99, "mem_gb_teacher": 9.99}
|
| 75 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.40700901261599115, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 7.09375, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.339, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.5, "frames": {"chat": 208}, "mem_gb": 9.99, "mem_gb_teacher": 9.99}
|
| 76 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3612343382894993, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 6.28125, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.588, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.5, "frames": {"chat": 221}, "mem_gb": 10.07, "mem_gb_teacher": 10.07}
|
| 77 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3821119113404304, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 6.90625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.479, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.9, "frames": {"chat": 232}, "mem_gb": 9.91, "mem_gb_teacher": 9.91}
|
| 78 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3988335593829552, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 1.796875, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.43, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.7, "frames": {"chat": 208}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 79 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.36860200558168194, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 3.09375, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.356, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.7, "frames": {"chat": 227}, "mem_gb": 9.86, "mem_gb_teacher": 9.86}
|
| 80 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.39240696791845064, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 1.9453125, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.387, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.5, "frames": {"chat": 221}, "mem_gb": 9.89, "mem_gb_teacher": 9.89}
|
| 81 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.38983683857706686, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 1.796875, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.457, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.8, "frames": {"chat": 232}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 82 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3839597153416524, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 1.390625, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.423, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.4, "frames": {"chat": 219}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 83 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.40005449274579685, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 1.015625, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.399, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.8, "frames": {"chat": 195}, "mem_gb": 10.05, "mem_gb_teacher": 10.05}
|
| 84 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.37729036068630717, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 2.421875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.503, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.9, "frames": {"chat": 216}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 85 |
+
{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.39155043416718643, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 1.640625, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.37, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.9, "frames": {"chat": 208}, "mem_gb": 9.84, "mem_gb_teacher": 9.84}
|
| 86 |
+
{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3371589448125412, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 1.625, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.407, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.5, "frames": {"chat": 235}, "mem_gb": 9.83, "mem_gb_teacher": 9.83}
|
| 87 |
+
{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3228179054065297, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 1.015625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.391, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.9, "frames": {"chat": 225}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 88 |
+
{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3611420020165543, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 2.078125, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.483, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.3, "frames": {"chat": 213}, "mem_gb": 10.03, "mem_gb_teacher": 10.03}
|
| 89 |
+
{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3108938364227613, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 1.25, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.372, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.4, "frames": {"chat": 257}, "mem_gb": 9.71, "mem_gb_teacher": 9.71}
|
| 90 |
+
{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3567947539317111, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 2.515625, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.497, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.9, "frames": {"chat": 212}, "mem_gb": 9.98, "mem_gb_teacher": 9.98}
|
| 91 |
+
{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.35618535781440636, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 2.65625, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.338, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.2, "frames": {"chat": 221}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 92 |
+
{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3354658261674146, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 2.390625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.442, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.0, "frames": {"chat": 242}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 93 |
+
{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3363398057249685, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 1.53125, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.352, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.6, "frames": {"chat": 220}, "mem_gb": 9.91, "mem_gb_teacher": 9.91}
|
| 94 |
+
{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30306674740935363, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.7734375, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.291, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.5, "frames": {"chat": 240}, "mem_gb": 9.81, "mem_gb_teacher": 9.81}
|
| 95 |
+
{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.29799054561704397, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 1.40625, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.359, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.8, "frames": {"chat": 206}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 96 |
+
{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3275977665552249, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 1.4921875, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.402, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.6, "frames": {"chat": 226}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 97 |
+
{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.34486291259291274, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 1.25, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.382, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.8, "frames": {"chat": 244}, "mem_gb": 9.74, "mem_gb_teacher": 9.74}
|
| 98 |
+
{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.31863660610305766, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 1.09375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.406, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.9, "frames": {"chat": 224}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 99 |
+
{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.313539165522034, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.99609375, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.319, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.9, "frames": {"chat": 271}, "mem_gb": 9.68, "mem_gb_teacher": 9.68}
|
| 100 |
+
{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30199915543012323, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.80078125, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.362, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.5, "frames": {"chat": 236}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 101 |
+
{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30484250679599745, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.75, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.306, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.5, "frames": {"chat": 232}, "mem_gb": 9.83, "mem_gb_teacher": 9.83}
|
| 102 |
+
{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30012307230867447, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 1.1640625, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.392, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.3, "frames": {"chat": 210}, "mem_gb": 9.89, "mem_gb_teacher": 9.89}
|
| 103 |
+
{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2981501909478257, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 1.2421875, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.431, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.0, "frames": {"chat": 217}, "mem_gb": 9.85, "mem_gb_teacher": 9.85}
|
| 104 |
+
{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.29460206581093373, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.83203125, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.38, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.1, "frames": {"chat": 224}, "mem_gb": 9.97, "mem_gb_teacher": 9.97}
|
| 105 |
+
{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3155513667286684, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.83203125, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.481, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.3, "frames": {"chat": 203}, "mem_gb": 9.82, "mem_gb_teacher": 9.82}
|
| 106 |
+
{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2852651763110111, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.75, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.326, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.8, "frames": {"chat": 239}, "mem_gb": 9.92, "mem_gb_teacher": 9.92}
|
| 107 |
+
{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2750922793724885, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.93359375, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.301, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.9, "frames": {"chat": 254}, "mem_gb": 9.83, "mem_gb_teacher": 9.83}
|
| 108 |
+
{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.27038282493477067, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.91796875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.421, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.4, "frames": {"chat": 241}, "mem_gb": 9.93, "mem_gb_teacher": 9.93}
|
| 109 |
+
{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.30530283329064645, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.8671875, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.419, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.2, "frames": {"chat": 213}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 110 |
+
{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2827595801195751, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.78125, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.641, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.6, "frames": {"chat": 221}, "mem_gb": 10.0, "mem_gb_teacher": 10.0}
|
| 111 |
+
{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.3139219396378845, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.9375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.334, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.6, "frames": {"chat": 196}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 112 |
+
{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.29290350563563405, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 1.2109375, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.427, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 28.5, "frames": {"chat": 270}, "mem_gb": 9.77, "mem_gb_teacher": 9.77}
|
| 113 |
+
{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2911341037095835, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 1.2578125, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.327, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.3, "frames": {"chat": 216}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 114 |
+
{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.30757556294202804, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.97265625, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.314, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.2, "frames": {"chat": 200}, "mem_gb": 9.91, "mem_gb_teacher": 9.91}
|
| 115 |
+
{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.27265539040267467, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 1.046875, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.406, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.9, "frames": {"chat": 206}, "mem_gb": 9.86, "mem_gb_teacher": 9.86}
|
| 116 |
+
{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.25661204309028884, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 1.0703125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.4, "frames": {"chat": 234}, "mem_gb": 9.9, "mem_gb_teacher": 9.9}
|
| 117 |
+
{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.28587267751296364, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 1.078125, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.325, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.5, "frames": {"chat": 215}, "mem_gb": 9.91, "mem_gb_teacher": 9.91}
|
| 118 |
+
{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24581487802788615, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.8515625, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.447, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.0, "frames": {"chat": 255}, "mem_gb": 9.89, "mem_gb_teacher": 9.89}
|
| 119 |
+
{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2630173706655701, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 1.0390625, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.377, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.9, "frames": {"chat": 250}, "mem_gb": 9.77, "mem_gb_teacher": 9.77}
|
| 120 |
+
{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.26195900368392466, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.85546875, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.525, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.8, "frames": {"chat": 242}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 121 |
+
{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.27887726591676476, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.98828125, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.517, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.5, "frames": {"chat": 199}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 122 |
+
{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.31573768441453576, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 1.15625, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.386, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.7, "frames": {"chat": 208}, "mem_gb": 9.99, "mem_gb_teacher": 9.99}
|
| 123 |
+
{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2738492341738194, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.859375, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.304, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.6, "frames": {"chat": 208}, "mem_gb": 9.92, "mem_gb_teacher": 9.92}
|
| 124 |
+
{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.3032253828023871, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 1.21875, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.473, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.8, "frames": {"chat": 209}, "mem_gb": 10.07, "mem_gb_teacher": 10.07}
|
| 125 |
+
{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.27386389810865125, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 1.5859375, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.321, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.6, "frames": {"chat": 235}, "mem_gb": 9.91, "mem_gb_teacher": 9.91}
|
| 126 |
+
{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2811740072357158, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 1.4453125, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.503, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.8, "frames": {"chat": 204}, "mem_gb": 9.9, "mem_gb_teacher": 9.9}
|
| 127 |
+
{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.32506899852765103, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 2.453125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.498, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.8, "frames": {"chat": 208}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 128 |
+
{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.27416869887411593, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 1.375, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.456, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.2, "frames": {"chat": 240}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 129 |
+
{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.27367745394359033, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 1.390625, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.425, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.4, "frames": {"chat": 246}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 130 |
+
{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.27279284177869556, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 1.1171875, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.299, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.1, "frames": {"chat": 243}, "mem_gb": 9.76, "mem_gb_teacher": 9.76}
|
| 131 |
+
{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2835182323958725, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.953125, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.449, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.8, "frames": {"chat": 208}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 132 |
+
{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2755484388658156, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.8203125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.352, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.3, "frames": {"chat": 219}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 133 |
+
{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.27247595444793504, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.796875, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.447, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.3, "frames": {"chat": 211}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 134 |
+
{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.252491025553147, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.283, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.0, "frames": {"chat": 232}, "mem_gb": 9.92, "mem_gb_teacher": 9.92}
|
| 135 |
+
{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2650780773670723, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.71875, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.403, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.3, "frames": {"chat": 214}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 136 |
+
{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2518176361516118, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.890625, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.395, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.8, "frames": {"chat": 226}, "mem_gb": 9.85, "mem_gb_teacher": 9.85}
|
| 137 |
+
{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.23503943474429348, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.75, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.404, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.4, "frames": {"chat": 210}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 138 |
+
{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2334536761138588, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.6796875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.436, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.4, "frames": {"chat": 218}, "mem_gb": 9.78, "mem_gb_teacher": 9.78}
|
| 139 |
+
{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2328599476976941, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.354, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.5, "frames": {"chat": 233}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 140 |
+
{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.26084051485359666, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.70703125, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.221, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.0, "frames": {"chat": 215}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 141 |
+
{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.26040056609710055, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.83203125, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.428, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.3, "frames": {"chat": 233}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 142 |
+
{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2547849820467333, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.91015625, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.353, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.2, "frames": {"chat": 209}, "mem_gb": 9.89, "mem_gb_teacher": 9.89}
|
| 143 |
+
{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2485178017048786, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.83203125, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.308, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 27.5, "frames": {"chat": 262}, "mem_gb": 9.82, "mem_gb_teacher": 9.82}
|
| 144 |
+
{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.25363729545498886, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.7578125, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.241, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.7, "frames": {"chat": 249}, "mem_gb": 9.91, "mem_gb_teacher": 9.91}
|
| 145 |
+
{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.29562477170241375, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.9453125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.364, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 26.4, "frames": {"chat": 227}, "mem_gb": 9.95, "mem_gb_teacher": 9.95}
|
| 146 |
+
{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.25183481702382365, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.98828125, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.356, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.5, "frames": {"chat": 221}, "mem_gb": 9.94, "mem_gb_teacher": 9.94}
|
| 147 |
+
{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.2600625433813781, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.89453125, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.473, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.4, "frames": {"chat": 234}, "mem_gb": 9.96, "mem_gb_teacher": 9.96}
|
| 148 |
+
{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.24899872875362636, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.7265625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.29, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.5, "frames": {"chat": 213}, "mem_gb": 9.9, "mem_gb_teacher": 9.9}
|
| 149 |
+
{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.22898227033279836, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.87890625, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.455, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 24.7, "frames": {"chat": 213}, "mem_gb": 9.84, "mem_gb_teacher": 9.84}
|
| 150 |
+
{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.23123409348068139, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.82421875, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.392, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 25.8, "frames": {"chat": 234}, "mem_gb": 9.87, "mem_gb_teacher": 9.87}
|
healed/knee0924/keep20.console.log
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
|
| 4 |
+
[14:23:29] pre-heal PPL = 164.8023
|
| 5 |
+
[14:24:16] step 1/50 KL/token 2.7994 45.8s (122880 tokens)
|
| 6 |
+
[14:26:43] step 5/50 KL/token 1.6533 36.7s (614400 tokens)
|
| 7 |
+
[14:29:46] step 10/50 KL/token 1.3937 36.7s (1228800 tokens)
|
| 8 |
+
[14:32:50] step 15/50 KL/token 1.2130 36.7s (1843200 tokens)
|
| 9 |
+
[14:35:54] step 20/50 KL/token 1.2127 36.7s (2457600 tokens)
|
| 10 |
+
[14:38:57] step 25/50 KL/token 1.1050 36.7s (3072000 tokens)
|
| 11 |
+
[14:42:00] step 30/50 KL/token 1.1799 36.7s (3686400 tokens)
|
| 12 |
+
[14:45:04] step 35/50 KL/token 1.0958 36.7s (4300800 tokens)
|
| 13 |
+
[14:48:07] step 40/50 KL/token 1.0824 36.7s (4915200 tokens)
|
| 14 |
+
[14:51:10] step 45/50 KL/token 1.0122 36.7s (5529600 tokens)
|
| 15 |
+
[14:54:14] step 50/50 KL/token 1.0871 36.7s (6144000 tokens)
|
| 16 |
+
[14:54:22] saved healed checkpoint -> outputs/healed/knee0924/keep20/step0050
|
| 17 |
+
[14:54:25] post-heal PPL = 33.1851 (pre 164.8023)
|
| 18 |
+
[14:54:25] wrote outputs/healed/knee0924/keep20/heal_result.json
|
healed/knee0924/keep25.console.log
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
|
| 4 |
+
[13:51:15] pre-heal PPL = 105.4488
|
| 5 |
+
[13:52:04] step 1/50 KL/token 2.3138 47.4s (122880 tokens)
|
| 6 |
+
[13:54:34] step 5/50 KL/token 1.3798 37.4s (614400 tokens)
|
| 7 |
+
[13:57:41] step 10/50 KL/token 1.1639 37.4s (1228800 tokens)
|
| 8 |
+
[14:00:48] step 15/50 KL/token 1.0206 37.4s (1843200 tokens)
|
| 9 |
+
[14:03:55] step 20/50 KL/token 1.0258 37.4s (2457600 tokens)
|
| 10 |
+
[14:07:02] step 25/50 KL/token 0.9372 37.4s (3072000 tokens)
|
| 11 |
+
[14:10:09] step 30/50 KL/token 1.0117 37.4s (3686400 tokens)
|
| 12 |
+
[14:13:16] step 35/50 KL/token 0.9334 37.4s (4300800 tokens)
|
| 13 |
+
[14:16:23] step 40/50 KL/token 0.9237 37.4s (4915200 tokens)
|
| 14 |
+
[14:19:30] step 45/50 KL/token 0.8614 37.3s (5529600 tokens)
|
| 15 |
+
[14:22:36] step 50/50 KL/token 0.9383 37.3s (6144000 tokens)
|
| 16 |
+
[14:22:44] saved healed checkpoint -> outputs/healed/knee0924/keep25/step0050
|
| 17 |
+
[14:22:47] post-heal PPL = 28.7085 (pre 105.4488)
|
| 18 |
+
[14:22:47] wrote outputs/healed/knee0924/keep25/heal_result.json
|
healed/knee0924/keep30.console.log
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
|
| 4 |
+
[13:18:21] pre-heal PPL = 69.0992
|
| 5 |
+
[13:19:12] step 1/50 KL/token 1.9134 48.5s (122880 tokens)
|
| 6 |
+
[13:21:45] step 5/50 KL/token 1.1476 38.1s (614400 tokens)
|
| 7 |
+
[13:24:55] step 10/50 KL/token 0.9709 38.1s (1228800 tokens)
|
| 8 |
+
[13:28:06] step 15/50 KL/token 0.8551 38.1s (1843200 tokens)
|
| 9 |
+
[13:31:16] step 20/50 KL/token 0.8761 38.1s (2457600 tokens)
|
| 10 |
+
[13:34:27] step 25/50 KL/token 0.8031 38.1s (3072000 tokens)
|
| 11 |
+
[13:37:37] step 30/50 KL/token 0.8738 38.1s (3686400 tokens)
|
| 12 |
+
[13:40:48] step 35/50 KL/token 0.8033 38.1s (4300800 tokens)
|
| 13 |
+
[13:43:58] step 40/50 KL/token 0.8020 38.1s (4915200 tokens)
|
| 14 |
+
[13:47:08] step 45/50 KL/token 0.7483 38.1s (5529600 tokens)
|
| 15 |
+
[13:50:19] step 50/50 KL/token 0.8231 38.0s (6144000 tokens)
|
| 16 |
+
[13:50:30] saved healed checkpoint -> outputs/healed/knee0924/keep30/step0050
|
| 17 |
+
[13:50:32] post-heal PPL = 25.7052 (pre 69.0992)
|
| 18 |
+
[13:50:32] wrote outputs/healed/knee0924/keep30/heal_result.json
|
healed/knee0924/keep40.console.log
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
[12:43:20] pre-heal PPL = 34.6561
|
| 6 |
+
[12:44:12] step 1/50 KL/token 1.2030 49.6s (122880 tokens)
|
| 7 |
+
[12:46:53] step 5/50 KL/token 0.7818 40.4s (614400 tokens)
|
| 8 |
+
[12:50:15] step 10/50 KL/token 0.6813 40.5s (1228800 tokens)
|
| 9 |
+
[12:53:39] step 15/50 KL/token 0.6109 40.4s (1843200 tokens)
|
| 10 |
+
[12:57:01] step 20/50 KL/token 0.6348 40.4s (2457600 tokens)
|
| 11 |
+
[13:00:23] step 25/50 KL/token 0.5769 40.3s (3072000 tokens)
|
| 12 |
+
[13:03:44] step 30/50 KL/token 0.6396 40.2s (3686400 tokens)
|
| 13 |
+
[13:07:06] step 35/50 KL/token 0.5842 40.3s (4300800 tokens)
|
| 14 |
+
[13:10:27] step 40/50 KL/token 0.5914 40.3s (4915200 tokens)
|
| 15 |
+
[13:13:48] step 45/50 KL/token 0.5476 40.3s (5529600 tokens)
|
| 16 |
+
[13:17:10] step 50/50 KL/token 0.6161 40.2s (6144000 tokens)
|
| 17 |
+
[13:17:25] saved healed checkpoint -> outputs/healed/knee0924/keep40/step0050
|
| 18 |
+
[13:17:28] post-heal PPL = 21.0572 (pre 34.6561)
|
| 19 |
+
[13:17:28] wrote outputs/healed/knee0924/keep40/heal_result.json
|
healed/knee0924/keep50.console.log
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
[12:07:19] pre-heal PPL = 25.5902
|
| 6 |
+
[12:08:13] step 1/50 KL/token 0.8699 52.5s (122880 tokens)
|
| 7 |
+
[12:11:01] step 5/50 KL/token 0.5701 42.0s (614400 tokens)
|
| 8 |
+
[12:14:31] step 10/50 KL/token 0.4897 42.0s (1228800 tokens)
|
| 9 |
+
[12:18:01] step 15/50 KL/token 0.4419 42.0s (1843200 tokens)
|
| 10 |
+
[12:21:31] step 20/50 KL/token 0.4637 42.0s (2457600 tokens)
|
| 11 |
+
[12:25:01] step 25/50 KL/token 0.4170 42.0s (3072000 tokens)
|
| 12 |
+
[12:28:31] step 30/50 KL/token 0.4739 42.1s (3686400 tokens)
|
| 13 |
+
[12:32:01] step 35/50 KL/token 0.4287 42.0s (4300800 tokens)
|
| 14 |
+
[12:35:32] step 40/50 KL/token 0.4449 42.1s (4915200 tokens)
|
| 15 |
+
[12:39:02] step 45/50 KL/token 0.4022 42.1s (5529600 tokens)
|
| 16 |
+
[12:42:32] step 50/50 KL/token 0.4655 42.0s (6144000 tokens)
|
| 17 |
+
[12:42:47] saved healed checkpoint -> outputs/healed/knee0924/keep50/step0050
|
| 18 |
+
[12:42:51] post-heal PPL = 18.2544 (pre 25.5902)
|
| 19 |
+
[12:42:51] wrote outputs/healed/knee0924/keep50/heal_result.json
|
healed/mixdistill_smoke/args.json
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"student": "outputs/pruned/glean-0125inst-math-keep50",
|
| 3 |
+
"teacher": "allenai/OLMoE-1B-7B-0125-Instruct",
|
| 4 |
+
"training_mode": "on-policy",
|
| 5 |
+
"kl_direction": "reverse",
|
| 6 |
+
"dataset": "allenai/Dolci-Instruct-RL",
|
| 7 |
+
"dataset_sources": null,
|
| 8 |
+
"max_difficulty": null,
|
| 9 |
+
"trajectories": "outputs/teacher_trajectories/dolci_math_curated.jsonl",
|
| 10 |
+
"trajectory_dataset": "allenai/Dolci-Instruct-RL",
|
| 11 |
+
"off_policy_frames": "chat",
|
| 12 |
+
"off_policy_max_seq_len": 2048,
|
| 13 |
+
"topk_targets": null,
|
| 14 |
+
"max_loss_tokens": null,
|
| 15 |
+
"loss_tokens_per_step": null,
|
| 16 |
+
"teacher_device": "cuda:0",
|
| 17 |
+
"student_device": "cuda:1",
|
| 18 |
+
"lr": 3e-05,
|
| 19 |
+
"optimizer": "adamw8bit",
|
| 20 |
+
"weight_decay": 0.1,
|
| 21 |
+
"epochs": 1,
|
| 22 |
+
"prompts_per_step": 64,
|
| 23 |
+
"group_size": 4,
|
| 24 |
+
"rollout_batch": 64,
|
| 25 |
+
"micro_batch": 2,
|
| 26 |
+
"max_new_tokens": 2048,
|
| 27 |
+
"max_prompt_len": 1024,
|
| 28 |
+
"warmup_steps": 10,
|
| 29 |
+
"max_grad_norm": 1.0,
|
| 30 |
+
"eval_every": 10,
|
| 31 |
+
"gsm8k_every": 0,
|
| 32 |
+
"gsm8k_n": 256,
|
| 33 |
+
"gsm8k_batch": 16,
|
| 34 |
+
"gsm8k_max_new_tokens": 512,
|
| 35 |
+
"gsm8k_frames": "chat",
|
| 36 |
+
"save_every": 1000,
|
| 37 |
+
"out_dir": "outputs/healed/mixdistill_smoke",
|
| 38 |
+
"sweep": 6,
|
| 39 |
+
"wandb": false,
|
| 40 |
+
"wandb_project": "glean-heal",
|
| 41 |
+
"wandb_run_name": null,
|
| 42 |
+
"wandb_run_id": null,
|
| 43 |
+
"wandb_resume": null,
|
| 44 |
+
"wandb_mode": "offline",
|
| 45 |
+
"no_wandb_sync": true,
|
| 46 |
+
"debug": false,
|
| 47 |
+
"resume_from": null,
|
| 48 |
+
"start_step": 0,
|
| 49 |
+
"no_grad_checkpointing": false,
|
| 50 |
+
"seed": 1223,
|
| 51 |
+
"no_teacher_overlap": false,
|
| 52 |
+
"sync_checkpoints": false,
|
| 53 |
+
"rollout_engine": "vllm",
|
| 54 |
+
"vllm_gpu": "2",
|
| 55 |
+
"vllm_port": 8377,
|
| 56 |
+
"vllm_refresh_every": 1,
|
| 57 |
+
"vllm_serve_bin": "vllm-plugin/.venv/bin/python",
|
| 58 |
+
"vllm_gpu_mem_util": 0.85,
|
| 59 |
+
"gold_mix_lambda": 0.5,
|
| 60 |
+
"gold_topk_targets": "outputs/teacher_trajectories/dolci_combined_top128",
|
| 61 |
+
"gold_mix_decay": 0.0,
|
| 62 |
+
"fast_teacher": true,
|
| 63 |
+
"reference_kl_beta": 0.0,
|
| 64 |
+
"drop_truncated_rollouts": false,
|
| 65 |
+
"vllm_max_model_len": null,
|
| 66 |
+
"vllm_refresh_mode": "reload",
|
| 67 |
+
"vllm_live_dir": null,
|
| 68 |
+
"resolved_kl_direction": "reverse"
|
| 69 |
+
}
|
healed/mixdistill_smoke/train_log.jsonl
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"step": 1, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6562024963381666, "tokens": 41373, "cumulative_loss_tokens": 41373, "grad_norm": 4.875, "lr": 6e-06, "finish_rate": 1.0, "comp_len": 646.5, "dropped_truncated": 0, "gold_loss": 0.2629, "gold_lambda": 0.5, "rep_ratio": 2.442, "t_data_s": 0.0, "t_rollout_s": 29.6, "t_step_s": 59.6, "t_refresh_s": 0.3, "mem_gb": 9.62}
|
| 2 |
+
{"step": 2, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.721973983016071, "tokens": 33593, "cumulative_loss_tokens": 74966, "grad_norm": 5.8125, "lr": 9e-06, "finish_rate": 1.0, "comp_len": 524.9, "dropped_truncated": 0, "gold_loss": 0.3084, "gold_lambda": 0.5, "rep_ratio": 2.28, "t_data_s": 0.0, "t_rollout_s": 24.4, "t_step_s": 47.0, "t_refresh_s": 0.3, "mem_gb": 9.46}
|
| 3 |
+
{"step": 3, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8109471330806736, "tokens": 45130, "cumulative_loss_tokens": 120096, "grad_norm": 5.9375, "lr": 1.2e-05, "finish_rate": 1.0, "comp_len": 705.2, "dropped_truncated": 0, "gold_loss": 0.2756, "gold_lambda": 0.5, "rep_ratio": 2.475, "t_data_s": 0.0, "t_rollout_s": 27.1, "t_step_s": 50.8, "t_refresh_s": 0.3, "mem_gb": 9.56}
|
| 4 |
+
{"step": 4, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.33583703860762965, "tokens": 46248, "cumulative_loss_tokens": 166344, "grad_norm": 1.828125, "lr": 1.5e-05, "finish_rate": 0.953, "comp_len": 722.6, "dropped_truncated": 0, "gold_loss": 0.2457, "gold_lambda": 0.5, "rep_ratio": 2.685, "t_data_s": 0.0, "t_rollout_s": 30.5, "t_step_s": 53.8, "t_refresh_s": 0.3, "mem_gb": 9.83}
|
| 5 |
+
{"step": 5, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8250801293169474, "tokens": 42255, "cumulative_loss_tokens": 208599, "grad_norm": 6.21875, "lr": 1.8e-05, "finish_rate": 0.984, "comp_len": 660.2, "dropped_truncated": 0, "gold_loss": 0.2722, "gold_lambda": 0.5, "rep_ratio": 2.503, "t_data_s": 0.0, "t_rollout_s": 30.5, "t_step_s": 54.9, "t_refresh_s": 0.3, "mem_gb": 9.57}
|
| 6 |
+
{"step": 6, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3196084560057643, "tokens": 47556, "cumulative_loss_tokens": 256155, "grad_norm": 1.375, "lr": 2.1e-05, "finish_rate": 1.0, "comp_len": 743.1, "dropped_truncated": 0, "gold_loss": 0.2262, "gold_lambda": 0.5, "rep_ratio": 2.542, "t_data_s": 0.0, "t_rollout_s": 25.7, "t_step_s": 38.8, "t_refresh_s": 0.0, "mem_gb": 9.51}
|
healed/mixdistill_smoke/vllm_server.log
ADDED
|
@@ -0,0 +1,446 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Skipping import of cpp extensions due to incompatible torch version 2.10.0+cu128 for torchao version 0.15.0 Please see https://github.com/pytorch/ao/issues/2919 for more info
|
| 2 |
+
WARNING 07-30 19:12:53 [registry.py:915] Model architecture OlmoeForCausalLM is already registered, and will be overwritten by the new model class glean_vllm.pruned_olmoe:PrunedOlmoeForCausalLM.
|
| 3 |
+
(APIServer pid=2011266) INFO 07-30 19:12:54 [utils.py:299]
|
| 4 |
+
(APIServer pid=2011266) INFO 07-30 19:12:54 [utils.py:299] █ █ █▄ ▄█
|
| 5 |
+
(APIServer pid=2011266) INFO 07-30 19:12:54 [utils.py:299] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.19.0
|
| 6 |
+
(APIServer pid=2011266) INFO 07-30 19:12:54 [utils.py:299] █▄█▀ █ █ █ █ model outputs/pruned/glean-0125inst-math-keep50
|
| 7 |
+
(APIServer pid=2011266) INFO 07-30 19:12:54 [utils.py:299] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀
|
| 8 |
+
(APIServer pid=2011266) INFO 07-30 19:12:54 [utils.py:299]
|
| 9 |
+
(APIServer pid=2011266) INFO 07-30 19:12:54 [utils.py:233] non-default args: {'model_tag': 'outputs/pruned/glean-0125inst-math-keep50', 'host': '127.0.0.1', 'port': 8377, 'model': 'outputs/pruned/glean-0125inst-math-keep50', 'max_model_len': 3200, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 10 |
+
(APIServer pid=2011266) INFO 07-30 19:13:02 [model.py:549] Resolved architecture: OlmoeForCausalLM
|
| 11 |
+
(APIServer pid=2011266) INFO 07-30 19:13:02 [model.py:1678] Using max model len 3200
|
| 12 |
+
(APIServer pid=2011266) INFO 07-30 19:13:02 [vllm.py:790] Asynchronous scheduling is enabled.
|
| 13 |
+
(APIServer pid=2011266) WARNING 07-30 19:13:02 [vllm.py:848] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 14 |
+
(APIServer pid=2011266) WARNING 07-30 19:13:02 [vllm.py:859] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 15 |
+
(APIServer pid=2011266) INFO 07-30 19:13:02 [vllm.py:1025] Cudagraph is disabled under eager mode
|
| 16 |
+
(APIServer pid=2011266) INFO 07-30 19:13:02 [compilation.py:290] Enabled custom fusions: norm_quant, act_quant
|
| 17 |
+
Skipping import of cpp extensions due to incompatible torch version 2.10.0+cu128 for torchao version 0.15.0 Please see https://github.com/pytorch/ao/issues/2919 for more info
|
| 18 |
+
(EngineCore pid=2011607) WARNING 07-30 19:13:10 [registry.py:915] Model architecture OlmoeForCausalLM is already registered, and will be overwritten by the new model class glean_vllm.pruned_olmoe:PrunedOlmoeForCausalLM.
|
| 19 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:10 [core.py:105] Initializing a V1 LLM engine (v0.19.0) with config: model='outputs/pruned/glean-0125inst-math-keep50', speculative_config=None, tokenizer='outputs/pruned/glean-0125inst-math-keep50', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=3200, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_images_per_batch': 0, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': True, 'static_all_moe_layers': []}
|
| 20 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:10 [parallel_state.py:1400] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:34027 backend=nccl
|
| 21 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:10 [parallel_state.py:1716] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 22 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:11 [gpu_model_runner.py:4735] Starting to load model outputs/pruned/glean-0125inst-math-keep50...
|
| 23 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:12 [cuda.py:334] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 24 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:12 [flash_attn.py:596] Using FlashAttention version 2
|
| 25 |
+
(EngineCore pid=2011607)
|
| 26 |
+
(EngineCore pid=2011607)
|
| 27 |
+
(EngineCore pid=2011607)
|
| 28 |
+
(EngineCore pid=2011607)
|
| 29 |
+
(EngineCore pid=2011607)
|
| 30 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:17 [default_loader.py:384] Loading weights took 4.75 seconds
|
| 31 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:17 [gpu_model_runner.py:4820] Model loading took 6.89 GiB memory and 5.340972 seconds
|
| 32 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:19 [gpu_worker.py:436] Available KV cache memory: 12.86 GiB
|
| 33 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:19 [kv_cache_utils.py:1319] GPU KV cache size: 105,344 tokens
|
| 34 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:19 [kv_cache_utils.py:1324] Maximum concurrency for 3,200 tokens per request: 32.92x
|
| 35 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:19 [core.py:283] init engine (profile, create kv cache, warmup model) took 1.67 seconds
|
| 36 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:19 [vllm.py:790] Asynchronous scheduling is enabled.
|
| 37 |
+
(EngineCore pid=2011607) WARNING 07-30 19:13:19 [vllm.py:848] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 38 |
+
(EngineCore pid=2011607) WARNING 07-30 19:13:19 [vllm.py:859] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 39 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:19 [vllm.py:1025] Cudagraph is disabled under eager mode
|
| 40 |
+
(EngineCore pid=2011607) INFO 07-30 19:13:19 [compilation.py:290] Enabled custom fusions: norm_quant, act_quant
|
| 41 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [api_server.py:590] Supported tasks: ['generate']
|
| 42 |
+
(APIServer pid=2011266) WARNING 07-30 19:13:19 [__init__.py:14] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 43 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [hf.py:314] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 44 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [api_server.py:594] Starting vLLM server on http://127.0.0.1:8377
|
| 45 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:37] Available routes are:
|
| 46 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
|
| 47 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /docs, Methods: GET, HEAD
|
| 48 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
|
| 49 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
|
| 50 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /sleep, Methods: POST
|
| 51 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 52 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 53 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 54 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 55 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 56 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 57 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 58 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 59 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /load, Methods: GET
|
| 60 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /version, Methods: GET
|
| 61 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /health, Methods: GET
|
| 62 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /metrics, Methods: GET
|
| 63 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /server_info, Methods: GET
|
| 64 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 65 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /ping, Methods: GET
|
| 66 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /ping, Methods: POST
|
| 67 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /invocations, Methods: POST
|
| 68 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 69 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 70 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 71 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 72 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 73 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 74 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 75 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 76 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 77 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /pause, Methods: POST
|
| 78 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /resume, Methods: POST
|
| 79 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 80 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 81 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 82 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 83 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 84 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 85 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 86 |
+
(APIServer pid=2011266) INFO 07-30 19:13:19 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 87 |
+
(APIServer pid=2011266) INFO: Started server process [2011266]
|
| 88 |
+
(APIServer pid=2011266) INFO: Waiting for application startup.
|
| 89 |
+
(APIServer pid=2011266) INFO: Application startup complete.
|
| 90 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:46468 - "GET /health HTTP/1.1" 200 OK
|
| 91 |
+
(APIServer pid=2011266) INFO 07-30 19:13:50 [loggers.py:259] Engine 000: Avg prompt throughput: 227.6 tokens/s, Avg generation throughput: 2577.7 tokens/s, Running: 31 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.4%, Prefix cache hit rate: 68.5%
|
| 92 |
+
(APIServer pid=2011266) INFO 07-30 19:14:00 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 1320.1 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.8%, Prefix cache hit rate: 68.5%
|
| 93 |
+
(APIServer pid=2011266) INFO 07-30 19:14:10 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 238.8 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 68.5%
|
| 94 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:56324 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 95 |
+
(APIServer pid=2011266) INFO 07-30 19:14:20 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 68.5%
|
| 96 |
+
(EngineCore pid=2011607) INFO 07-30 19:14:39 [gpu_model_runner.py:4957] Reloading weights inplace...
|
| 97 |
+
(EngineCore pid=2011607)
|
| 98 |
+
(EngineCore pid=2011607)
|
| 99 |
+
(EngineCore pid=2011607)
|
| 100 |
+
(EngineCore pid=2011607)
|
| 101 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeForCausalLM: Failed to load weights
|
| 102 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeModel: Failed to load weights
|
| 103 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] ModuleList: Failed to load weights
|
| 104 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 105 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 106 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] RotaryEmbedding: Failed to load weights
|
| 107 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] ApplyRotaryEmb: Failed to load weights
|
| 108 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 109 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 110 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 111 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 112 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 113 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 114 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 115 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 116 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 117 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 118 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 119 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 120 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 121 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 122 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 123 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 124 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 125 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 126 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 127 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 128 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 129 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 130 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 131 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 132 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 133 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 134 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 135 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 136 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 137 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 138 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 139 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 140 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 141 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 142 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 143 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 144 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 145 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 146 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 147 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 148 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 149 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 150 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 151 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 152 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 153 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 154 |
+
(EngineCore pid=2011607) WARNING 07-30 19:14:40 [layerwise.py:230] LogitsProcessor: Failed to load weights
|
| 155 |
+
(EngineCore pid=2011607) INFO 07-30 19:14:40 [gpu_model_runner.py:4980] Reloading and processing weights took 0.31 seconds
|
| 156 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:55810 - "POST /collective_rpc HTTP/1.1" 200 OK
|
| 157 |
+
(APIServer pid=2011266) INFO 07-30 19:14:40 [api_router.py:39] Resetting prefix cache...
|
| 158 |
+
(EngineCore pid=2011607) INFO 07-30 19:14:40 [block_pool.py:472] Successfully reset prefix cache
|
| 159 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:55822 - "POST /reset_prefix_cache HTTP/1.1" 200 OK
|
| 160 |
+
(APIServer pid=2011266) INFO 07-30 19:14:50 [loggers.py:259] Engine 000: Avg prompt throughput: 230.8 tokens/s, Avg generation throughput: 2655.2 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.9%, Prefix cache hit rate: 69.6%
|
| 161 |
+
(APIServer pid=2011266) INFO 07-30 19:15:00 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 638.5 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 69.6%
|
| 162 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:40698 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 163 |
+
(APIServer pid=2011266) INFO 07-30 19:15:10 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 65.2 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 69.6%
|
| 164 |
+
(APIServer pid=2011266) INFO 07-30 19:15:20 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 69.6%
|
| 165 |
+
(EngineCore pid=2011607)
|
| 166 |
+
(EngineCore pid=2011607)
|
| 167 |
+
(EngineCore pid=2011607)
|
| 168 |
+
(EngineCore pid=2011607)
|
| 169 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeForCausalLM: Failed to load weights
|
| 170 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeModel: Failed to load weights
|
| 171 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] ModuleList: Failed to load weights
|
| 172 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 173 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 174 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] RotaryEmbedding: Failed to load weights
|
| 175 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] ApplyRotaryEmb: Failed to load weights
|
| 176 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 177 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 178 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 179 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 180 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 181 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 182 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 183 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 184 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 185 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 186 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 187 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 188 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 189 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 190 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 191 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 192 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 193 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 194 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 195 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 196 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 197 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 198 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 199 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 200 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 201 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 202 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 203 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 204 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 205 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 206 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 207 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 208 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 209 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 210 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 211 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 212 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 213 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 214 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 215 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 216 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 217 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 218 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 219 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 220 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 221 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 222 |
+
(EngineCore pid=2011607) WARNING 07-30 19:15:27 [layerwise.py:230] LogitsProcessor: Failed to load weights
|
| 223 |
+
(EngineCore pid=2011607) INFO 07-30 19:15:27 [gpu_model_runner.py:4980] Reloading and processing weights took 0.29 seconds
|
| 224 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:57242 - "POST /collective_rpc HTTP/1.1" 200 OK
|
| 225 |
+
(APIServer pid=2011266) INFO 07-30 19:15:27 [api_router.py:39] Resetting prefix cache...
|
| 226 |
+
(EngineCore pid=2011607) INFO 07-30 19:15:27 [block_pool.py:472] Successfully reset prefix cache
|
| 227 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:57252 - "POST /reset_prefix_cache HTTP/1.1" 200 OK
|
| 228 |
+
(APIServer pid=2011266) INFO 07-30 19:15:30 [loggers.py:259] Engine 000: Avg prompt throughput: 321.5 tokens/s, Avg generation throughput: 908.1 tokens/s, Running: 60 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.8%, Prefix cache hit rate: 71.2%
|
| 229 |
+
(APIServer pid=2011266) INFO 07-30 19:15:40 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 2574.3 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.5%, Prefix cache hit rate: 71.2%
|
| 230 |
+
(APIServer pid=2011266) INFO 07-30 19:15:50 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 976.5 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.9%, Prefix cache hit rate: 71.2%
|
| 231 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:57264 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 232 |
+
(APIServer pid=2011266) INFO 07-30 19:16:00 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 53.7 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 71.2%
|
| 233 |
+
(APIServer pid=2011266) INFO 07-30 19:16:10 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 71.2%
|
| 234 |
+
(EngineCore pid=2011607)
|
| 235 |
+
(EngineCore pid=2011607)
|
| 236 |
+
(EngineCore pid=2011607)
|
| 237 |
+
(EngineCore pid=2011607)
|
| 238 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeForCausalLM: Failed to load weights
|
| 239 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeModel: Failed to load weights
|
| 240 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] ModuleList: Failed to load weights
|
| 241 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 242 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 243 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] RotaryEmbedding: Failed to load weights
|
| 244 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] ApplyRotaryEmb: Failed to load weights
|
| 245 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 246 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 247 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 248 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 249 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 250 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 251 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 252 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 253 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 254 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 255 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 256 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 257 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 258 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 259 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 260 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 261 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 262 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 263 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 264 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 265 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 266 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 267 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 268 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 269 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 270 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 271 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 272 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 273 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 274 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 275 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 276 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 277 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 278 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 279 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 280 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 281 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 282 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 283 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 284 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 285 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 286 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 287 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 288 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 289 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 290 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 291 |
+
(EngineCore pid=2011607) WARNING 07-30 19:16:18 [layerwise.py:230] LogitsProcessor: Failed to load weights
|
| 292 |
+
(EngineCore pid=2011607) INFO 07-30 19:16:18 [gpu_model_runner.py:4980] Reloading and processing weights took 0.29 seconds
|
| 293 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:51270 - "POST /collective_rpc HTTP/1.1" 200 OK
|
| 294 |
+
(APIServer pid=2011266) INFO 07-30 19:16:18 [api_router.py:39] Resetting prefix cache...
|
| 295 |
+
(EngineCore pid=2011607) INFO 07-30 19:16:18 [block_pool.py:472] Successfully reset prefix cache
|
| 296 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:51274 - "POST /reset_prefix_cache HTTP/1.1" 200 OK
|
| 297 |
+
(APIServer pid=2011266) INFO 07-30 19:16:20 [loggers.py:259] Engine 000: Avg prompt throughput: 210.8 tokens/s, Avg generation throughput: 620.7 tokens/s, Running: 62 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 67.4%
|
| 298 |
+
(APIServer pid=2011266) INFO 07-30 19:16:30 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 2937.0 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 18.1%, Prefix cache hit rate: 67.4%
|
| 299 |
+
(APIServer pid=2011266) INFO 07-30 19:16:40 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 871.4 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.8%, Prefix cache hit rate: 67.4%
|
| 300 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:51284 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 301 |
+
(APIServer pid=2011266) INFO 07-30 19:16:50 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 195.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 67.4%
|
| 302 |
+
(APIServer pid=2011266) INFO 07-30 19:17:00 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 67.4%
|
| 303 |
+
(EngineCore pid=2011607)
|
| 304 |
+
(EngineCore pid=2011607)
|
| 305 |
+
(EngineCore pid=2011607)
|
| 306 |
+
(EngineCore pid=2011607)
|
| 307 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeForCausalLM: Failed to load weights
|
| 308 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeModel: Failed to load weights
|
| 309 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] ModuleList: Failed to load weights
|
| 310 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 311 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 312 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] RotaryEmbedding: Failed to load weights
|
| 313 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] ApplyRotaryEmb: Failed to load weights
|
| 314 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 315 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 316 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 317 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 318 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 319 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 320 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 321 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 322 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 323 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 324 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 325 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 326 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 327 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 328 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 329 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 330 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 331 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 332 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 333 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 334 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 335 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 336 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 337 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 338 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 339 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 340 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 341 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 342 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 343 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 344 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 345 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 346 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 347 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 348 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 349 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 350 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 351 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 352 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 353 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 354 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 355 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 356 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 357 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 358 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 359 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 360 |
+
(EngineCore pid=2011607) WARNING 07-30 19:17:12 [layerwise.py:230] LogitsProcessor: Failed to load weights
|
| 361 |
+
(EngineCore pid=2011607) INFO 07-30 19:17:12 [gpu_model_runner.py:4980] Reloading and processing weights took 0.28 seconds
|
| 362 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:42974 - "POST /collective_rpc HTTP/1.1" 200 OK
|
| 363 |
+
(APIServer pid=2011266) INFO 07-30 19:17:12 [api_router.py:39] Resetting prefix cache...
|
| 364 |
+
(EngineCore pid=2011607) INFO 07-30 19:17:12 [block_pool.py:472] Successfully reset prefix cache
|
| 365 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:42980 - "POST /reset_prefix_cache HTTP/1.1" 200 OK
|
| 366 |
+
(APIServer pid=2011266) INFO 07-30 19:17:20 [loggers.py:259] Engine 000: Avg prompt throughput: 271.6 tokens/s, Avg generation throughput: 2593.0 tokens/s, Running: 43 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.6%, Prefix cache hit rate: 69.8%
|
| 367 |
+
(APIServer pid=2011266) INFO 07-30 19:17:30 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 1420.2 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.9%, Prefix cache hit rate: 69.8%
|
| 368 |
+
(APIServer pid=2011266) INFO 07-30 19:17:40 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 192.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 69.8%
|
| 369 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:42984 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 370 |
+
(APIServer pid=2011266) INFO 07-30 19:17:50 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 19.1 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 69.8%
|
| 371 |
+
(APIServer pid=2011266) INFO 07-30 19:18:00 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 69.8%
|
| 372 |
+
(EngineCore pid=2011607)
|
| 373 |
+
(EngineCore pid=2011607)
|
| 374 |
+
(EngineCore pid=2011607)
|
| 375 |
+
(EngineCore pid=2011607)
|
| 376 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeForCausalLM: Failed to load weights
|
| 377 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeModel: Failed to load weights
|
| 378 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] ModuleList: Failed to load weights
|
| 379 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 380 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 381 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] RotaryEmbedding: Failed to load weights
|
| 382 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] ApplyRotaryEmb: Failed to load weights
|
| 383 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 384 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 385 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 386 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 387 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 388 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 389 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 390 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 391 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 392 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 393 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 394 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 395 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 396 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 397 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 398 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 399 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 400 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 401 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 402 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 403 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 404 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 405 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 406 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 407 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 408 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 409 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 410 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 411 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 412 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 413 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 414 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 415 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 416 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 417 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 418 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 419 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 420 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 421 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 422 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 423 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 424 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 425 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 426 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] PrunedOlmoeDecoderLayer: Failed to load weights
|
| 427 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] OlmoeAttention: Failed to load weights
|
| 428 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] VllmVariableOlmoeMoE: Failed to load weights
|
| 429 |
+
(EngineCore pid=2011607) WARNING 07-30 19:18:07 [layerwise.py:230] LogitsProcessor: Failed to load weights
|
| 430 |
+
(EngineCore pid=2011607) INFO 07-30 19:18:07 [gpu_model_runner.py:4980] Reloading and processing weights took 0.29 seconds
|
| 431 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:45206 - "POST /collective_rpc HTTP/1.1" 200 OK
|
| 432 |
+
(APIServer pid=2011266) INFO 07-30 19:18:07 [api_router.py:39] Resetting prefix cache...
|
| 433 |
+
(EngineCore pid=2011607) INFO 07-30 19:18:07 [block_pool.py:472] Successfully reset prefix cache
|
| 434 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:45222 - "POST /reset_prefix_cache HTTP/1.1" 200 OK
|
| 435 |
+
(APIServer pid=2011266) INFO 07-30 19:18:10 [loggers.py:259] Engine 000: Avg prompt throughput: 267.6 tokens/s, Avg generation throughput: 1109.1 tokens/s, Running: 59 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.7%, Prefix cache hit rate: 70.0%
|
| 436 |
+
(APIServer pid=2011266) INFO 07-30 19:18:20 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 2707.2 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.9%, Prefix cache hit rate: 70.0%
|
| 437 |
+
(APIServer pid=2011266) INFO 07-30 19:18:30 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 902.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 70.0%
|
| 438 |
+
(APIServer pid=2011266) INFO: 127.0.0.1:45232 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 439 |
+
(APIServer pid=2011266) INFO 07-30 19:18:40 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 36.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 70.0%
|
| 440 |
+
(APIServer pid=2011266) INFO 07-30 19:18:50 [loggers.py:259] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 70.0%
|
| 441 |
+
(EngineCore pid=2011607) INFO 07-30 19:18:58 [core.py:1210] Shutdown initiated (timeout=0)
|
| 442 |
+
(EngineCore pid=2011607) INFO 07-30 19:18:58 [core.py:1233] Shutdown complete
|
| 443 |
+
(APIServer pid=2011266) INFO: Shutting down
|
| 444 |
+
(APIServer pid=2011266) INFO: Waiting for application shutdown.
|
| 445 |
+
(APIServer pid=2011266) INFO: Application shutdown complete.
|
| 446 |
+
(APIServer pid=2011266) INFO: Finished server process [2011266]
|
healed/soak2048_filtered_keep50_s1223/args.json
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"student": "outputs/pruned/glean-0125inst-math-keep50",
|
| 3 |
+
"teacher": "allenai/OLMoE-1B-7B-0125-Instruct",
|
| 4 |
+
"training_mode": "on-policy",
|
| 5 |
+
"kl_direction": "reverse",
|
| 6 |
+
"dataset": "allenai/Dolci-Instruct-RL",
|
| 7 |
+
"dataset_sources": null,
|
| 8 |
+
"max_difficulty": null,
|
| 9 |
+
"trajectories": "outputs/teacher_trajectories/dolci_math_curated.jsonl",
|
| 10 |
+
"trajectory_dataset": "allenai/Dolci-Instruct-RL",
|
| 11 |
+
"off_policy_frames": "chat",
|
| 12 |
+
"off_policy_max_seq_len": 2048,
|
| 13 |
+
"topk_targets": null,
|
| 14 |
+
"max_loss_tokens": null,
|
| 15 |
+
"loss_tokens_per_step": null,
|
| 16 |
+
"teacher_device": "cuda:0",
|
| 17 |
+
"student_device": "cuda:1",
|
| 18 |
+
"lr": 3e-05,
|
| 19 |
+
"optimizer": "adamw8bit",
|
| 20 |
+
"weight_decay": 0.1,
|
| 21 |
+
"epochs": 2,
|
| 22 |
+
"prompts_per_step": 256,
|
| 23 |
+
"group_size": 4,
|
| 24 |
+
"rollout_batch": 64,
|
| 25 |
+
"micro_batch": 4,
|
| 26 |
+
"max_new_tokens": 2048,
|
| 27 |
+
"max_prompt_len": 1024,
|
| 28 |
+
"warmup_steps": 10,
|
| 29 |
+
"max_grad_norm": 1.0,
|
| 30 |
+
"eval_every": 10,
|
| 31 |
+
"gsm8k_every": 25,
|
| 32 |
+
"gsm8k_n": 256,
|
| 33 |
+
"gsm8k_batch": 16,
|
| 34 |
+
"gsm8k_max_new_tokens": 1024,
|
| 35 |
+
"gsm8k_frames": "chat",
|
| 36 |
+
"save_every": 25,
|
| 37 |
+
"out_dir": "outputs/healed/soak2048_filtered_keep50_s1223",
|
| 38 |
+
"sweep": 150,
|
| 39 |
+
"wandb": true,
|
| 40 |
+
"wandb_project": "glean-heal",
|
| 41 |
+
"wandb_run_name": "soak2048-filtered-keep50-s1223",
|
| 42 |
+
"wandb_run_id": null,
|
| 43 |
+
"wandb_resume": null,
|
| 44 |
+
"wandb_mode": "offline",
|
| 45 |
+
"no_wandb_sync": false,
|
| 46 |
+
"debug": false,
|
| 47 |
+
"resume_from": null,
|
| 48 |
+
"start_step": 0,
|
| 49 |
+
"no_grad_checkpointing": false,
|
| 50 |
+
"seed": 1223,
|
| 51 |
+
"no_teacher_overlap": false,
|
| 52 |
+
"sync_checkpoints": false,
|
| 53 |
+
"rollout_engine": "vllm",
|
| 54 |
+
"vllm_gpu": "2",
|
| 55 |
+
"vllm_port": 8377,
|
| 56 |
+
"vllm_refresh_every": 1,
|
| 57 |
+
"vllm_serve_bin": "vllm-plugin/.venv/bin/python",
|
| 58 |
+
"vllm_gpu_mem_util": 0.85,
|
| 59 |
+
"drop_truncated_rollouts": true,
|
| 60 |
+
"vllm_max_model_len": null,
|
| 61 |
+
"vllm_refresh_mode": "reload",
|
| 62 |
+
"vllm_live_dir": null,
|
| 63 |
+
"resolved_kl_direction": "reverse"
|
| 64 |
+
}
|
healed/soak2048_filtered_keep50_s1223/train_log.jsonl
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"step": 1, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6384803785151999, "tokens": 160226, "cumulative_loss_tokens": 160226, "grad_norm": 4.09375, "lr": 6e-06, "finish_rate": 0.977, "comp_len": 673.9, "dropped_truncated": 6, "t_data_s": 0.0, "t_rollout_s": 51.7, "t_step_s": 114.0, "t_refresh_s": 0.3, "mem_gb": 10.09}
|
| 2 |
+
{"step": 1, "gsm8k_n": 256, "gsm8k_quick_chat": 0.5703125, "t_eval_s": 23.9}
|
| 3 |
+
{"step": 2, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5822258179742169, "tokens": 171907, "cumulative_loss_tokens": 332133, "grad_norm": 4.15625, "lr": 9e-06, "finish_rate": 0.98, "comp_len": 711.5, "dropped_truncated": 5, "t_data_s": 0.0, "t_rollout_s": 55.2, "t_step_s": 113.4, "t_refresh_s": 0.3, "mem_gb": 10.1}
|
| 4 |
+
{"step": 3, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6279944151936665, "tokens": 165922, "cumulative_loss_tokens": 498055, "grad_norm": 3.5625, "lr": 1.2e-05, "finish_rate": 0.977, "comp_len": 696.1, "dropped_truncated": 6, "t_data_s": 0.0, "t_rollout_s": 53.5, "t_step_s": 111.1, "t_refresh_s": 0.3, "mem_gb": 10.13}
|
| 5 |
+
{"step": 4, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.541299577955838, "tokens": 156764, "cumulative_loss_tokens": 654819, "grad_norm": 3.28125, "lr": 1.5e-05, "finish_rate": 0.984, "comp_len": 644.4, "dropped_truncated": 4, "t_data_s": 0.0, "t_rollout_s": 49.3, "t_step_s": 103.5, "t_refresh_s": 0.3, "mem_gb": 10.08}
|
| 6 |
+
{"step": 5, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5785494116399411, "tokens": 158290, "cumulative_loss_tokens": 813109, "grad_norm": 3.484375, "lr": 1.8e-05, "finish_rate": 0.977, "comp_len": 666.3, "dropped_truncated": 6, "t_data_s": 0.0, "t_rollout_s": 51.1, "t_step_s": 106.3, "t_refresh_s": 0.3, "mem_gb": 10.06}
|
| 7 |
+
{"step": 6, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.49468118802685496, "tokens": 152259, "cumulative_loss_tokens": 965368, "grad_norm": 2.40625, "lr": 2.1e-05, "finish_rate": 0.973, "comp_len": 650.8, "dropped_truncated": 7, "t_data_s": 0.0, "t_rollout_s": 52.7, "t_step_s": 109.5, "t_refresh_s": 0.3, "mem_gb": 10.42}
|
| 8 |
+
{"step": 7, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4172215888346972, "tokens": 172176, "cumulative_loss_tokens": 1137544, "grad_norm": 1.7109375, "lr": 2.4e-05, "finish_rate": 0.988, "comp_len": 696.6, "dropped_truncated": 3, "t_data_s": 0.0, "t_rollout_s": 54.0, "t_step_s": 112.0, "t_refresh_s": 0.3, "mem_gb": 10.08}
|
| 9 |
+
{"step": 8, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.47372317872629416, "tokens": 164732, "cumulative_loss_tokens": 1302276, "grad_norm": 1.609375, "lr": 2.7000000000000002e-05, "finish_rate": 0.984, "comp_len": 675.5, "dropped_truncated": 4, "t_data_s": 0.0, "t_rollout_s": 51.1, "t_step_s": 107.9, "t_refresh_s": 0.3, "mem_gb": 10.21}
|
| 10 |
+
{"step": 9, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6217017395292128, "tokens": 165138, "cumulative_loss_tokens": 1467414, "grad_norm": 1.65625, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 693.1, "dropped_truncated": 6, "t_data_s": 0.0, "t_rollout_s": 53.6, "t_step_s": 111.9, "t_refresh_s": 0.3, "mem_gb": 10.06}
|
| 11 |
+
{"step": 10, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.508857000384844, "tokens": 177450, "cumulative_loss_tokens": 1644864, "grad_norm": 0.9140625, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 733.2, "dropped_truncated": 5, "t_data_s": 0.0, "t_rollout_s": 55.3, "t_step_s": 112.9, "t_refresh_s": 0.3, "mem_gb": 10.15}
|
| 12 |
+
{"step": 11, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4898551779152319, "tokens": 170656, "cumulative_loss_tokens": 1815520, "grad_norm": 1.921875, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 706.6, "dropped_truncated": 5, "t_data_s": 0.0, "t_rollout_s": 53.1, "t_step_s": 110.7, "t_refresh_s": 0.3, "mem_gb": 10.12}
|
| 13 |
+
{"step": 12, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5149587567519402, "tokens": 201003, "cumulative_loss_tokens": 2016523, "grad_norm": 2.578125, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 849.2, "dropped_truncated": 8, "t_data_s": 0.0, "t_rollout_s": 64.2, "t_step_s": 128.5, "t_refresh_s": 0.3, "mem_gb": 10.12}
|
| 14 |
+
{"step": 13, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6249667176931957, "tokens": 177694, "cumulative_loss_tokens": 2194217, "grad_norm": 2.484375, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 742.1, "dropped_truncated": 6, "t_data_s": 0.0, "t_rollout_s": 55.1, "t_step_s": 112.8, "t_refresh_s": 0.3, "mem_gb": 10.07}
|
| 15 |
+
{"step": 14, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6021736298644345, "tokens": 183128, "cumulative_loss_tokens": 2377345, "grad_norm": 3.328125, "lr": 3e-05, "finish_rate": 0.984, "comp_len": 747.3, "dropped_truncated": 4, "t_data_s": 0.0, "t_rollout_s": 55.3, "t_step_s": 115.3, "t_refresh_s": 0.3, "mem_gb": 10.15}
|
| 16 |
+
{"step": 15, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6021683271185623, "tokens": 184776, "cumulative_loss_tokens": 2562121, "grad_norm": 1.3828125, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 785.8, "dropped_truncated": 8, "t_data_s": 0.0, "t_rollout_s": 57.7, "t_step_s": 117.8, "t_refresh_s": 0.3, "mem_gb": 10.18}
|
| 17 |
+
{"step": 16, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5942752304250699, "tokens": 192733, "cumulative_loss_tokens": 2754854, "grad_norm": 1.1953125, "lr": 3e-05, "finish_rate": 0.957, "comp_len": 840.9, "dropped_truncated": 11, "t_data_s": 0.0, "t_rollout_s": 62.1, "t_step_s": 122.6, "t_refresh_s": 0.3, "mem_gb": 10.13}
|
| 18 |
+
{"step": 17, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.695920866143972, "tokens": 180925, "cumulative_loss_tokens": 2935779, "grad_norm": 1.390625, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 754.7, "dropped_truncated": 6, "t_data_s": 0.0, "t_rollout_s": 56.4, "t_step_s": 116.0, "t_refresh_s": 0.3, "mem_gb": 10.19}
|
| 19 |
+
{"step": 18, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6393542171970616, "tokens": 176091, "cumulative_loss_tokens": 3111870, "grad_norm": 1.359375, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 751.9, "dropped_truncated": 8, "t_data_s": 0.0, "t_rollout_s": 58.6, "t_step_s": 120.7, "t_refresh_s": 0.3, "mem_gb": 10.19}
|
| 20 |
+
{"step": 19, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6169990370698945, "tokens": 174262, "cumulative_loss_tokens": 3286132, "grad_norm": 1.0703125, "lr": 3e-05, "finish_rate": 0.957, "comp_len": 768.7, "dropped_truncated": 11, "t_data_s": 0.0, "t_rollout_s": 58.0, "t_step_s": 116.5, "t_refresh_s": 0.3, "mem_gb": 10.2}
|
| 21 |
+
{"step": 20, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5433280002840769, "tokens": 187315, "cumulative_loss_tokens": 3473447, "grad_norm": 1.203125, "lr": 3e-05, "finish_rate": 0.961, "comp_len": 811.7, "dropped_truncated": 10, "t_data_s": 0.0, "t_rollout_s": 60.8, "t_step_s": 121.8, "t_refresh_s": 0.3, "mem_gb": 10.06}
|
| 22 |
+
{"step": 21, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.651670906829498, "tokens": 159821, "cumulative_loss_tokens": 3633268, "grad_norm": 1.234375, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 648.3, "dropped_truncated": 3, "t_data_s": 0.0, "t_rollout_s": 49.9, "t_step_s": 106.2, "t_refresh_s": 0.3, "mem_gb": 10.27}
|
| 23 |
+
{"step": 22, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6278111477779328, "tokens": 177630, "cumulative_loss_tokens": 3810898, "grad_norm": 1.3515625, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 733.9, "dropped_truncated": 5, "t_data_s": 0.0, "t_rollout_s": 55.7, "t_step_s": 117.4, "t_refresh_s": 0.3, "mem_gb": 10.24}
|
| 24 |
+
{"step": 23, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.724997552869678, "tokens": 140330, "cumulative_loss_tokens": 3951228, "grad_norm": 1.6328125, "lr": 3e-05, "finish_rate": 1.0, "comp_len": 548.2, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 39.9, "t_step_s": 89.9, "t_refresh_s": 0.3, "mem_gb": 10.03}
|
| 25 |
+
{"step": 24, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6501219767447058, "tokens": 142815, "cumulative_loss_tokens": 4094043, "grad_norm": 1.7109375, "lr": 3e-05, "finish_rate": 1.0, "comp_len": 557.9, "dropped_truncated": 0, "t_data_s": 0.0, "t_rollout_s": 38.0, "t_step_s": 89.5, "t_refresh_s": 0.3, "mem_gb": 9.88}
|
| 26 |
+
{"step": 25, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6675267936828272, "tokens": 161887, "cumulative_loss_tokens": 4255930, "grad_norm": 1.7109375, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 680.4, "dropped_truncated": 6, "t_data_s": 0.0, "t_rollout_s": 51.8, "t_step_s": 108.8, "t_refresh_s": 0.3, "mem_gb": 10.07}
|
| 27 |
+
{"step": 25, "gsm8k_n": 256, "gsm8k_quick_chat": 0.578125, "t_eval_s": 15.0}
|
| 28 |
+
{"step": 26, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6354966885301152, "tokens": 156177, "cumulative_loss_tokens": 4412107, "grad_norm": 2.40625, "lr": 3e-05, "finish_rate": 0.996, "comp_len": 618.1, "dropped_truncated": 1, "t_data_s": 0.0, "t_rollout_s": 49.2, "t_step_s": 103.2, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 29 |
+
{"step": 27, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7035677946763135, "tokens": 157103, "cumulative_loss_tokens": 4569210, "grad_norm": 2.421875, "lr": 3e-05, "finish_rate": 0.992, "comp_len": 629.7, "dropped_truncated": 2, "t_data_s": 0.0, "t_rollout_s": 48.7, "t_step_s": 102.5, "t_refresh_s": 0.3, "mem_gb": 10.06}
|
| 30 |
+
{"step": 28, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7211097005633599, "tokens": 160688, "cumulative_loss_tokens": 4729898, "grad_norm": 2.40625, "lr": 3e-05, "finish_rate": 0.996, "comp_len": 635.7, "dropped_truncated": 1, "t_data_s": 0.0, "t_rollout_s": 48.6, "t_step_s": 102.7, "t_refresh_s": 0.3, "mem_gb": 10.11}
|
| 31 |
+
{"step": 29, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7847257256680407, "tokens": 143369, "cumulative_loss_tokens": 4873267, "grad_norm": 1.703125, "lr": 3e-05, "finish_rate": 0.957, "comp_len": 648.0, "dropped_truncated": 11, "t_data_s": 0.0, "t_rollout_s": 49.7, "t_step_s": 101.5, "t_refresh_s": 0.3, "mem_gb": 10.01}
|
| 32 |
+
{"step": 30, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.699323502096092, "tokens": 188622, "cumulative_loss_tokens": 5061889, "grad_norm": 3.953125, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 784.8, "dropped_truncated": 6, "t_data_s": 0.0, "t_rollout_s": 58.4, "t_step_s": 121.4, "t_refresh_s": 0.3, "mem_gb": 10.13}
|
| 33 |
+
{"step": 31, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6853019188045449, "tokens": 168202, "cumulative_loss_tokens": 5230091, "grad_norm": 5.15625, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 929.0, "dropped_truncated": 34, "t_data_s": 0.0, "t_rollout_s": 74.3, "t_step_s": 132.0, "t_refresh_s": 0.3, "mem_gb": 10.18}
|
| 34 |
+
{"step": 32, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.757249608680153, "tokens": 169046, "cumulative_loss_tokens": 5399137, "grad_norm": 8.6875, "lr": 3e-05, "finish_rate": 0.953, "comp_len": 756.3, "dropped_truncated": 12, "t_data_s": 0.0, "t_rollout_s": 57.5, "t_step_s": 115.4, "t_refresh_s": 0.3, "mem_gb": 10.08}
|
| 35 |
+
{"step": 33, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.895772915829715, "tokens": 156022, "cumulative_loss_tokens": 5555159, "grad_norm": 8.8125, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 857.5, "dropped_truncated": 31, "t_data_s": 0.0, "t_rollout_s": 66.2, "t_step_s": 122.0, "t_refresh_s": 0.3, "mem_gb": 10.17}
|
| 36 |
+
{"step": 34, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7674467855807574, "tokens": 164594, "cumulative_loss_tokens": 5719753, "grad_norm": 4.6875, "lr": 3e-05, "finish_rate": 0.898, "comp_len": 850.9, "dropped_truncated": 26, "t_data_s": 0.0, "t_rollout_s": 66.3, "t_step_s": 124.3, "t_refresh_s": 0.3, "mem_gb": 10.35}
|
| 37 |
+
{"step": 35, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8783435842807807, "tokens": 153378, "cumulative_loss_tokens": 5873131, "grad_norm": 4.40625, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 879.1, "dropped_truncated": 35, "t_data_s": 0.0, "t_rollout_s": 67.7, "t_step_s": 121.2, "t_refresh_s": 0.3, "mem_gb": 10.1}
|
| 38 |
+
{"step": 36, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.9287989634230446, "tokens": 137780, "cumulative_loss_tokens": 6010911, "grad_norm": 2.5, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 962.2, "dropped_truncated": 53, "t_data_s": 0.0, "t_rollout_s": 77.4, "t_step_s": 127.3, "t_refresh_s": 0.3, "mem_gb": 10.07}
|
| 39 |
+
{"step": 37, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8785660122888517, "tokens": 153849, "cumulative_loss_tokens": 6164760, "grad_norm": 2.546875, "lr": 3e-05, "finish_rate": 0.812, "comp_len": 985.0, "dropped_truncated": 48, "t_data_s": 0.0, "t_rollout_s": 80.0, "t_step_s": 130.9, "t_refresh_s": 0.3, "mem_gb": 10.17}
|
| 40 |
+
{"step": 38, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7734455308228608, "tokens": 144701, "cumulative_loss_tokens": 6309461, "grad_norm": 4.28125, "lr": 3e-05, "finish_rate": 0.645, "comp_len": 1293.2, "dropped_truncated": 91, "t_data_s": 0.0, "t_rollout_s": 121.0, "t_step_s": 169.4, "t_refresh_s": 0.3, "mem_gb": 10.11}
|
| 41 |
+
{"step": 39, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7804501209626984, "tokens": 127733, "cumulative_loss_tokens": 6437194, "grad_norm": 8.375, "lr": 3e-05, "finish_rate": 0.605, "comp_len": 1307.0, "dropped_truncated": 101, "t_data_s": 0.0, "t_rollout_s": 121.5, "t_step_s": 166.4, "t_refresh_s": 0.3, "mem_gb": 10.35}
|
| 42 |
+
{"step": 40, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8111260535034754, "tokens": 138161, "cumulative_loss_tokens": 6575355, "grad_norm": 2.84375, "lr": 3e-05, "finish_rate": 0.664, "comp_len": 1227.7, "dropped_truncated": 86, "t_data_s": 0.0, "t_rollout_s": 114.0, "t_step_s": 160.1, "t_refresh_s": 0.3, "mem_gb": 10.03}
|
| 43 |
+
{"step": 41, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8330551704190711, "tokens": 146779, "cumulative_loss_tokens": 6722134, "grad_norm": 2.578125, "lr": 3e-05, "finish_rate": 0.727, "comp_len": 1133.4, "dropped_truncated": 70, "t_data_s": 0.0, "t_rollout_s": 105.4, "t_step_s": 154.9, "t_refresh_s": 0.3, "mem_gb": 10.04}
|
| 44 |
+
{"step": 42, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8665060712802937, "tokens": 162785, "cumulative_loss_tokens": 6884919, "grad_norm": 2.53125, "lr": 3e-05, "finish_rate": 0.848, "comp_len": 947.9, "dropped_truncated": 39, "t_data_s": 0.0, "t_rollout_s": 75.6, "t_step_s": 129.8, "t_refresh_s": 0.3, "mem_gb": 10.09}
|
| 45 |
+
{"step": 43, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7808148046784279, "tokens": 159142, "cumulative_loss_tokens": 7044061, "grad_norm": 2.9375, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 901.6, "dropped_truncated": 35, "t_data_s": 0.0, "t_rollout_s": 67.8, "t_step_s": 121.6, "t_refresh_s": 0.3, "mem_gb": 10.08}
|
| 46 |
+
{"step": 44, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.9077712043682102, "tokens": 180319, "cumulative_loss_tokens": 7224380, "grad_norm": 2.9375, "lr": 3e-05, "finish_rate": 0.941, "comp_len": 824.4, "dropped_truncated": 15, "t_data_s": 0.0, "t_rollout_s": 62.2, "t_step_s": 121.3, "t_refresh_s": 0.3, "mem_gb": 10.16}
|
| 47 |
+
{"step": 45, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8197376252935459, "tokens": 151831, "cumulative_loss_tokens": 7376211, "grad_norm": 1.546875, "lr": 3e-05, "finish_rate": 0.91, "comp_len": 777.1, "dropped_truncated": 23, "t_data_s": 0.0, "t_rollout_s": 58.2, "t_step_s": 110.3, "t_refresh_s": 0.3, "mem_gb": 10.13}
|
| 48 |
+
{"step": 46, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7152417010924011, "tokens": 165462, "cumulative_loss_tokens": 7541673, "grad_norm": 2.25, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 718.3, "dropped_truncated": 9, "t_data_s": 0.0, "t_rollout_s": 52.8, "t_step_s": 108.6, "t_refresh_s": 0.3, "mem_gb": 10.21}
|
| 49 |
+
{"step": 47, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7891334552247886, "tokens": 159720, "cumulative_loss_tokens": 7701393, "grad_norm": 2.28125, "lr": 3e-05, "finish_rate": 0.945, "comp_len": 735.9, "dropped_truncated": 14, "t_data_s": 0.0, "t_rollout_s": 55.2, "t_step_s": 113.2, "t_refresh_s": 0.3, "mem_gb": 10.04}
|
| 50 |
+
{"step": 48, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7440059168233365, "tokens": 163574, "cumulative_loss_tokens": 7864967, "grad_norm": 1.984375, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 679.0, "dropped_truncated": 5, "t_data_s": 0.0, "t_rollout_s": 50.6, "t_step_s": 105.3, "t_refresh_s": 0.3, "mem_gb": 10.08}
|
| 51 |
+
{"step": 49, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7736463516277142, "tokens": 157051, "cumulative_loss_tokens": 8022018, "grad_norm": 2.25, "lr": 3e-05, "finish_rate": 0.957, "comp_len": 701.5, "dropped_truncated": 11, "t_data_s": 0.0, "t_rollout_s": 52.5, "t_step_s": 106.7, "t_refresh_s": 0.3, "mem_gb": 10.14}
|
| 52 |
+
{"step": 50, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7645250728030133, "tokens": 145824, "cumulative_loss_tokens": 8167842, "grad_norm": 1.75, "lr": 3e-05, "finish_rate": 0.941, "comp_len": 689.6, "dropped_truncated": 15, "t_data_s": 0.0, "t_rollout_s": 52.7, "t_step_s": 105.0, "t_refresh_s": 0.3, "mem_gb": 10.08}
|
| 53 |
+
{"step": 50, "gsm8k_n": 256, "gsm8k_quick_chat": 0.5, "t_eval_s": 22.0}
|
| 54 |
+
{"step": 51, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7361227881185582, "tokens": 140973, "cumulative_loss_tokens": 8308815, "grad_norm": 2.0625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 750.7, "dropped_truncated": 25, "t_data_s": 0.0, "t_rollout_s": 56.8, "t_step_s": 106.0, "t_refresh_s": 0.3, "mem_gb": 9.95}
|
| 55 |
+
{"step": 52, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7424841233116772, "tokens": 141691, "cumulative_loss_tokens": 8450506, "grad_norm": 2.171875, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 729.5, "dropped_truncated": 22, "t_data_s": 0.0, "t_rollout_s": 55.3, "t_step_s": 105.2, "t_refresh_s": 0.3, "mem_gb": 9.96}
|
| 56 |
+
{"step": 53, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7469882032901451, "tokens": 156366, "cumulative_loss_tokens": 8606872, "grad_norm": 1.734375, "lr": 3e-05, "finish_rate": 0.91, "comp_len": 794.8, "dropped_truncated": 23, "t_data_s": 0.0, "t_rollout_s": 61.0, "t_step_s": 115.6, "t_refresh_s": 0.3, "mem_gb": 10.13}
|
| 57 |
+
{"step": 54, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8122103749000273, "tokens": 176478, "cumulative_loss_tokens": 8783350, "grad_norm": 4.0625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 889.4, "dropped_truncated": 25, "t_data_s": 0.0, "t_rollout_s": 70.5, "t_step_s": 129.6, "t_refresh_s": 0.3, "mem_gb": 10.27}
|
| 58 |
+
{"step": 55, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8376631480372229, "tokens": 163220, "cumulative_loss_tokens": 8946570, "grad_norm": 2.125, "lr": 3e-05, "finish_rate": 0.875, "comp_len": 893.6, "dropped_truncated": 32, "t_data_s": 0.0, "t_rollout_s": 67.6, "t_step_s": 122.5, "t_refresh_s": 0.3, "mem_gb": 10.28}
|
| 59 |
+
{"step": 56, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7692607203891041, "tokens": 170816, "cumulative_loss_tokens": 9117386, "grad_norm": 2.078125, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 907.2, "dropped_truncated": 30, "t_data_s": 0.0, "t_rollout_s": 70.9, "t_step_s": 129.2, "t_refresh_s": 0.3, "mem_gb": 10.26}
|
| 60 |
+
{"step": 57, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.841661863303352, "tokens": 163650, "cumulative_loss_tokens": 9281036, "grad_norm": 2.828125, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 887.3, "dropped_truncated": 31, "t_data_s": 0.0, "t_rollout_s": 68.1, "t_step_s": 121.3, "t_refresh_s": 0.3, "mem_gb": 10.07}
|
| 61 |
+
{"step": 58, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7292739853432862, "tokens": 175611, "cumulative_loss_tokens": 9456647, "grad_norm": 1.5703125, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 886.0, "dropped_truncated": 25, "t_data_s": 0.0, "t_rollout_s": 69.4, "t_step_s": 125.1, "t_refresh_s": 0.3, "mem_gb": 10.21}
|
| 62 |
+
{"step": 59, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7852089271316834, "tokens": 172867, "cumulative_loss_tokens": 9629514, "grad_norm": 2.109375, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 947.3, "dropped_truncated": 34, "t_data_s": 0.0, "t_rollout_s": 78.2, "t_step_s": 135.6, "t_refresh_s": 0.3, "mem_gb": 10.1}
|
| 63 |
+
{"step": 60, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8180076730868334, "tokens": 161458, "cumulative_loss_tokens": 9790972, "grad_norm": 1.9609375, "lr": 3e-05, "finish_rate": 0.895, "comp_len": 846.7, "dropped_truncated": 27, "t_data_s": 0.0, "t_rollout_s": 63.7, "t_step_s": 119.7, "t_refresh_s": 0.3, "mem_gb": 10.07}
|
| 64 |
+
{"step": 61, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7899970788064339, "tokens": 170722, "cumulative_loss_tokens": 9961694, "grad_norm": 1.921875, "lr": 3e-05, "finish_rate": 0.918, "comp_len": 834.9, "dropped_truncated": 21, "t_data_s": 0.0, "t_rollout_s": 62.3, "t_step_s": 119.1, "t_refresh_s": 0.3, "mem_gb": 10.09}
|
| 65 |
+
{"step": 62, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7264108320599143, "tokens": 186925, "cumulative_loss_tokens": 10148619, "grad_norm": 2.140625, "lr": 3e-05, "finish_rate": 0.91, "comp_len": 914.2, "dropped_truncated": 23, "t_data_s": 0.0, "t_rollout_s": 74.6, "t_step_s": 136.2, "t_refresh_s": 0.3, "mem_gb": 10.11}
|
| 66 |
+
{"step": 63, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7585512034482911, "tokens": 168150, "cumulative_loss_tokens": 10316769, "grad_norm": 4.375, "lr": 3e-05, "finish_rate": 0.898, "comp_len": 864.8, "dropped_truncated": 26, "t_data_s": 0.0, "t_rollout_s": 64.7, "t_step_s": 121.4, "t_refresh_s": 0.3, "mem_gb": 9.99}
|
| 67 |
+
{"step": 64, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7844444387865807, "tokens": 176519, "cumulative_loss_tokens": 10493288, "grad_norm": 2.296875, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 849.5, "dropped_truncated": 20, "t_data_s": 0.0, "t_rollout_s": 66.4, "t_step_s": 126.3, "t_refresh_s": 0.3, "mem_gb": 10.31}
|
| 68 |
+
{"step": 65, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7913634151858838, "tokens": 158506, "cumulative_loss_tokens": 10651794, "grad_norm": 1.9921875, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 811.2, "dropped_truncated": 24, "t_data_s": 0.0, "t_rollout_s": 60.8, "t_step_s": 114.3, "t_refresh_s": 0.3, "mem_gb": 10.07}
|
| 69 |
+
{"step": 66, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7712098291913507, "tokens": 183907, "cumulative_loss_tokens": 10835701, "grad_norm": 2.046875, "lr": 3e-05, "finish_rate": 0.953, "comp_len": 814.4, "dropped_truncated": 12, "t_data_s": 0.0, "t_rollout_s": 60.9, "t_step_s": 120.4, "t_refresh_s": 0.3, "mem_gb": 10.12}
|
| 70 |
+
{"step": 67, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6951695225434065, "tokens": 171294, "cumulative_loss_tokens": 11006995, "grad_norm": 1.7890625, "lr": 3e-05, "finish_rate": 0.938, "comp_len": 797.1, "dropped_truncated": 16, "t_data_s": 0.0, "t_rollout_s": 59.0, "t_step_s": 115.6, "t_refresh_s": 0.3, "mem_gb": 10.19}
|
| 71 |
+
{"step": 68, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7640094187486476, "tokens": 171359, "cumulative_loss_tokens": 11178354, "grad_norm": 2.546875, "lr": 3e-05, "finish_rate": 0.949, "comp_len": 773.4, "dropped_truncated": 13, "t_data_s": 0.0, "t_rollout_s": 56.6, "t_step_s": 111.6, "t_refresh_s": 0.3, "mem_gb": 10.08}
|
| 72 |
+
{"step": 69, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7263362626893161, "tokens": 174100, "cumulative_loss_tokens": 11352454, "grad_norm": 2.46875, "lr": 3e-05, "finish_rate": 0.957, "comp_len": 768.1, "dropped_truncated": 11, "t_data_s": 0.0, "t_rollout_s": 56.7, "t_step_s": 115.8, "t_refresh_s": 0.3, "mem_gb": 10.02}
|
| 73 |
+
{"step": 70, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7891206572589142, "tokens": 167028, "cumulative_loss_tokens": 11519482, "grad_norm": 1.3671875, "lr": 3e-05, "finish_rate": 0.93, "comp_len": 796.5, "dropped_truncated": 18, "t_data_s": 0.0, "t_rollout_s": 59.8, "t_step_s": 116.8, "t_refresh_s": 0.3, "mem_gb": 10.08}
|
| 74 |
+
{"step": 71, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8260723305393537, "tokens": 152532, "cumulative_loss_tokens": 11672014, "grad_norm": 2.03125, "lr": 3e-05, "finish_rate": 0.934, "comp_len": 731.8, "dropped_truncated": 17, "t_data_s": 0.0, "t_rollout_s": 54.5, "t_step_s": 107.2, "t_refresh_s": 0.3, "mem_gb": 10.09}
|
| 75 |
+
{"step": 72, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.690422797811594, "tokens": 178237, "cumulative_loss_tokens": 11850251, "grad_norm": 1.2578125, "lr": 3e-05, "finish_rate": 0.945, "comp_len": 808.2, "dropped_truncated": 14, "t_data_s": 0.0, "t_rollout_s": 60.6, "t_step_s": 120.5, "t_refresh_s": 0.3, "mem_gb": 10.12}
|
| 76 |
+
{"step": 73, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6913638442019767, "tokens": 170669, "cumulative_loss_tokens": 12020920, "grad_norm": 1.2578125, "lr": 3e-05, "finish_rate": 0.953, "comp_len": 762.7, "dropped_truncated": 12, "t_data_s": 0.0, "t_rollout_s": 56.5, "t_step_s": 112.5, "t_refresh_s": 0.3, "mem_gb": 10.15}
|
| 77 |
+
{"step": 74, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7147183445384437, "tokens": 171929, "cumulative_loss_tokens": 12192849, "grad_norm": 1.3203125, "lr": 3e-05, "finish_rate": 0.93, "comp_len": 815.6, "dropped_truncated": 18, "t_data_s": 0.0, "t_rollout_s": 60.9, "t_step_s": 118.0, "t_refresh_s": 0.3, "mem_gb": 10.18}
|
| 78 |
+
{"step": 75, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7489180493985259, "tokens": 164067, "cumulative_loss_tokens": 12356916, "grad_norm": 1.59375, "lr": 3e-05, "finish_rate": 0.957, "comp_len": 728.9, "dropped_truncated": 11, "t_data_s": 0.0, "t_rollout_s": 54.2, "t_step_s": 110.2, "t_refresh_s": 0.3, "mem_gb": 10.14}
|
| 79 |
+
{"step": 75, "gsm8k_n": 256, "gsm8k_quick_chat": 0.5078125, "t_eval_s": 17.1}
|
| 80 |
+
{"step": 76, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.664618537845775, "tokens": 188438, "cumulative_loss_tokens": 12545354, "grad_norm": 1.453125, "lr": 3e-05, "finish_rate": 0.961, "comp_len": 816.1, "dropped_truncated": 10, "t_data_s": 0.0, "t_rollout_s": 59.2, "t_step_s": 118.8, "t_refresh_s": 0.3, "mem_gb": 10.08}
|
| 81 |
+
{"step": 77, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.718582398728061, "tokens": 179688, "cumulative_loss_tokens": 12725042, "grad_norm": 1.4296875, "lr": 3e-05, "finish_rate": 0.949, "comp_len": 805.9, "dropped_truncated": 13, "t_data_s": 0.0, "t_rollout_s": 61.2, "t_step_s": 120.8, "t_refresh_s": 0.3, "mem_gb": 10.28}
|
| 82 |
+
{"step": 78, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8081445555764152, "tokens": 156885, "cumulative_loss_tokens": 12881927, "grad_norm": 1.6953125, "lr": 3e-05, "finish_rate": 0.941, "comp_len": 732.8, "dropped_truncated": 15, "t_data_s": 0.0, "t_rollout_s": 56.1, "t_step_s": 113.4, "t_refresh_s": 0.3, "mem_gb": 10.06}
|
| 83 |
+
{"step": 79, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7617091419785587, "tokens": 158715, "cumulative_loss_tokens": 13040642, "grad_norm": 3.375, "lr": 3e-05, "finish_rate": 0.918, "comp_len": 788.0, "dropped_truncated": 21, "t_data_s": 0.0, "t_rollout_s": 58.6, "t_step_s": 112.4, "t_refresh_s": 0.3, "mem_gb": 10.03}
|
| 84 |
+
{"step": 80, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8082229503603449, "tokens": 169549, "cumulative_loss_tokens": 13210191, "grad_norm": 4.4375, "lr": 3e-05, "finish_rate": 0.934, "comp_len": 798.3, "dropped_truncated": 17, "t_data_s": 0.0, "t_rollout_s": 59.7, "t_step_s": 117.1, "t_refresh_s": 0.3, "mem_gb": 10.26}
|
| 85 |
+
{"step": 81, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.729323075272184, "tokens": 187865, "cumulative_loss_tokens": 13398056, "grad_norm": 4.96875, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 893.8, "dropped_truncated": 20, "t_data_s": 0.0, "t_rollout_s": 71.2, "t_step_s": 133.0, "t_refresh_s": 0.3, "mem_gb": 10.19}
|
| 86 |
+
{"step": 82, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7518282090349567, "tokens": 180499, "cumulative_loss_tokens": 13578555, "grad_norm": 2.875, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 857.1, "dropped_truncated": 19, "t_data_s": 0.0, "t_rollout_s": 65.8, "t_step_s": 125.1, "t_refresh_s": 0.3, "mem_gb": 10.21}
|
| 87 |
+
{"step": 83, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7320501278707717, "tokens": 188650, "cumulative_loss_tokens": 13767205, "grad_norm": 2.75, "lr": 3e-05, "finish_rate": 0.953, "comp_len": 832.9, "dropped_truncated": 12, "t_data_s": 0.0, "t_rollout_s": 63.2, "t_step_s": 123.5, "t_refresh_s": 0.3, "mem_gb": 10.31}
|
| 88 |
+
{"step": 84, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7755936597314583, "tokens": 167776, "cumulative_loss_tokens": 13934981, "grad_norm": 2.5625, "lr": 3e-05, "finish_rate": 0.934, "comp_len": 791.4, "dropped_truncated": 17, "t_data_s": 0.0, "t_rollout_s": 59.3, "t_step_s": 115.6, "t_refresh_s": 0.3, "mem_gb": 10.16}
|
| 89 |
+
{"step": 85, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.682906216161391, "tokens": 174196, "cumulative_loss_tokens": 14109177, "grad_norm": 1.515625, "lr": 3e-05, "finish_rate": 0.949, "comp_len": 784.5, "dropped_truncated": 13, "t_data_s": 0.0, "t_rollout_s": 59.1, "t_step_s": 116.1, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 90 |
+
{"step": 86, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8049343342097134, "tokens": 156765, "cumulative_loss_tokens": 14265942, "grad_norm": 2.875, "lr": 3e-05, "finish_rate": 0.945, "comp_len": 724.4, "dropped_truncated": 14, "t_data_s": 0.0, "t_rollout_s": 54.7, "t_step_s": 108.9, "t_refresh_s": 0.3, "mem_gb": 10.14}
|
| 91 |
+
{"step": 87, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.9445644556196507, "tokens": 145898, "cumulative_loss_tokens": 14411840, "grad_norm": 4.4375, "lr": 3e-05, "finish_rate": 0.938, "comp_len": 697.9, "dropped_truncated": 16, "t_data_s": 0.0, "t_rollout_s": 52.0, "t_step_s": 104.8, "t_refresh_s": 0.3, "mem_gb": 10.07}
|
| 92 |
+
{"step": 88, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.9444579714959463, "tokens": 156372, "cumulative_loss_tokens": 14568212, "grad_norm": 6.6875, "lr": 3e-05, "finish_rate": 0.883, "comp_len": 850.8, "dropped_truncated": 30, "t_data_s": 0.0, "t_rollout_s": 65.6, "t_step_s": 121.3, "t_refresh_s": 0.3, "mem_gb": 10.17}
|
| 93 |
+
{"step": 89, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.2070773954291933, "tokens": 143896, "cumulative_loss_tokens": 14712108, "grad_norm": 8.5, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 1034.1, "dropped_truncated": 59, "t_data_s": 0.0, "t_rollout_s": 100.8, "t_step_s": 155.6, "t_refresh_s": 0.3, "mem_gb": 10.12}
|
| 94 |
+
{"step": 90, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.4058866115372848, "tokens": 100807, "cumulative_loss_tokens": 14812915, "grad_norm": 8.125, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 817.8, "dropped_truncated": 53, "t_data_s": 0.0, "t_rollout_s": 83.6, "t_step_s": 133.7, "t_refresh_s": 0.3, "mem_gb": 10.09}
|
| 95 |
+
{"step": 91, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.883159250457844, "tokens": 43319, "cumulative_loss_tokens": 14856234, "grad_norm": 16.625, "lr": 3e-05, "finish_rate": 0.777, "comp_len": 625.2, "dropped_truncated": 57, "t_data_s": 0.0, "t_rollout_s": 63.0, "t_step_s": 101.8, "t_refresh_s": 0.3, "mem_gb": 10.06}
|
| 96 |
+
{"step": 92, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.431357851099147, "tokens": 51255, "cumulative_loss_tokens": 14907489, "grad_norm": 27.125, "lr": 3e-05, "finish_rate": 0.703, "comp_len": 808.2, "dropped_truncated": 76, "t_data_s": 0.0, "t_rollout_s": 80.8, "t_step_s": 120.8, "t_refresh_s": 0.3, "mem_gb": 10.05}
|
| 97 |
+
{"step": 93, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.2755716946527003, "tokens": 64941, "cumulative_loss_tokens": 14972430, "grad_norm": 23.875, "lr": 3e-05, "finish_rate": 0.73, "comp_len": 805.7, "dropped_truncated": 69, "t_data_s": 0.0, "t_rollout_s": 78.0, "t_step_s": 122.5, "t_refresh_s": 0.3, "mem_gb": 9.92}
|
| 98 |
+
{"step": 94, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.022895620045153, "tokens": 108231, "cumulative_loss_tokens": 15080661, "grad_norm": 29.25, "lr": 3e-05, "finish_rate": 0.676, "comp_len": 1086.8, "dropped_truncated": 83, "t_data_s": 0.0, "t_rollout_s": 100.1, "t_step_s": 146.6, "t_refresh_s": 0.3, "mem_gb": 10.12}
|
| 99 |
+
{"step": 95, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.9414970366613401, "tokens": 69458, "cumulative_loss_tokens": 15150119, "grad_norm": 56.25, "lr": 3e-05, "finish_rate": 0.41, "comp_len": 1479.3, "dropped_truncated": 151, "t_data_s": 0.0, "t_rollout_s": 142.3, "t_step_s": 176.6, "t_refresh_s": 0.3, "mem_gb": 10.07}
|
| 100 |
+
{"step": 96, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.2904777843410247, "tokens": 15310, "cumulative_loss_tokens": 15165429, "grad_norm": 113.0, "lr": 3e-05, "finish_rate": 0.094, "comp_len": 1915.8, "dropped_truncated": 232, "t_data_s": 0.0, "t_rollout_s": 189.6, "t_step_s": 208.1, "t_refresh_s": 0.3, "mem_gb": 10.02}
|
| 101 |
+
{"step": 97, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7204363758974848, "tokens": 3830, "cumulative_loss_tokens": 15169259, "grad_norm": 125.0, "lr": 3e-05, "finish_rate": 0.043, "comp_len": 1975.0, "dropped_truncated": 245, "t_data_s": 0.0, "t_rollout_s": 196.0, "t_step_s": 209.1, "t_refresh_s": 0.3, "mem_gb": 9.78}
|
| 102 |
+
{"step": 98, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.9783305864295233, "tokens": 1472, "cumulative_loss_tokens": 15170731, "grad_norm": 80.5, "lr": 3e-05, "finish_rate": 0.043, "comp_len": 1965.8, "dropped_truncated": 245, "t_data_s": 0.0, "t_rollout_s": 192.0, "t_step_s": 204.1, "t_refresh_s": 0.3, "mem_gb": 8.48}
|
| 103 |
+
{"step": 99, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.4492854377286295, "tokens": 8528, "cumulative_loss_tokens": 15179259, "grad_norm": 442.0, "lr": 3e-05, "finish_rate": 0.086, "comp_len": 1905.3, "dropped_truncated": 234, "t_data_s": 0.0, "t_rollout_s": 186.5, "t_step_s": 201.0, "t_refresh_s": 0.3, "mem_gb": 9.61}
|
| 104 |
+
{"step": 100, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 2.776938802588368, "tokens": 66171, "cumulative_loss_tokens": 15245430, "grad_norm": 600.0, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 322.5, "dropped_truncated": 8, "t_data_s": 0.0, "t_rollout_s": 33.1, "t_step_s": 73.4, "t_refresh_s": 0.3, "mem_gb": 9.85}
|
| 105 |
+
{"step": 100, "gsm8k_n": 256, "gsm8k_quick_chat": 0.00390625, "t_eval_s": 18.6}
|
healed/soak2048_filtered_keep50_s1223/vllm_server.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/soak2048_filtered_keep50_s1223/wandb_sync.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Find logs at: /tmp/debug-cli.henry.log
|
| 2 |
+
Syncing: https://wandb.ai/hbfreed/glean-heal/runs/xjkc9if6 ... done.
|
healed/stableopd_cold_keep50/args.json
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"student": "outputs/pruned/glean-0125inst-math-keep50",
|
| 3 |
+
"teacher": "allenai/OLMoE-1B-7B-0125-Instruct",
|
| 4 |
+
"training_mode": "on-policy",
|
| 5 |
+
"kl_direction": "reverse",
|
| 6 |
+
"dataset": "allenai/Dolci-Instruct-RL",
|
| 7 |
+
"dataset_sources": null,
|
| 8 |
+
"max_difficulty": null,
|
| 9 |
+
"trajectories": "outputs/teacher_trajectories/dolci_math_curated.jsonl",
|
| 10 |
+
"trajectory_dataset": "allenai/Dolci-Instruct-RL",
|
| 11 |
+
"off_policy_frames": "chat",
|
| 12 |
+
"off_policy_max_seq_len": 2048,
|
| 13 |
+
"topk_targets": null,
|
| 14 |
+
"max_loss_tokens": null,
|
| 15 |
+
"loss_tokens_per_step": null,
|
| 16 |
+
"teacher_device": "cuda:0",
|
| 17 |
+
"student_device": "cuda:1",
|
| 18 |
+
"lr": 3e-06,
|
| 19 |
+
"optimizer": "adamw8bit",
|
| 20 |
+
"weight_decay": 0.1,
|
| 21 |
+
"epochs": 2,
|
| 22 |
+
"prompts_per_step": 256,
|
| 23 |
+
"group_size": 4,
|
| 24 |
+
"rollout_batch": 64,
|
| 25 |
+
"micro_batch": 4,
|
| 26 |
+
"max_new_tokens": 2048,
|
| 27 |
+
"max_prompt_len": 1024,
|
| 28 |
+
"warmup_steps": 10,
|
| 29 |
+
"max_grad_norm": 1.0,
|
| 30 |
+
"eval_every": 10,
|
| 31 |
+
"gsm8k_every": 20,
|
| 32 |
+
"gsm8k_n": 256,
|
| 33 |
+
"gsm8k_batch": 16,
|
| 34 |
+
"gsm8k_max_new_tokens": 1024,
|
| 35 |
+
"gsm8k_frames": "chat",
|
| 36 |
+
"save_every": 1000,
|
| 37 |
+
"out_dir": "outputs/healed/stableopd_cold_keep50",
|
| 38 |
+
"sweep": 120,
|
| 39 |
+
"wandb": true,
|
| 40 |
+
"wandb_project": "glean-heal",
|
| 41 |
+
"wandb_run_name": "stableopd-cold-keep50-s1223",
|
| 42 |
+
"wandb_run_id": null,
|
| 43 |
+
"wandb_resume": null,
|
| 44 |
+
"wandb_mode": "offline",
|
| 45 |
+
"no_wandb_sync": false,
|
| 46 |
+
"debug": false,
|
| 47 |
+
"resume_from": null,
|
| 48 |
+
"start_step": 0,
|
| 49 |
+
"no_grad_checkpointing": false,
|
| 50 |
+
"seed": 1223,
|
| 51 |
+
"no_teacher_overlap": false,
|
| 52 |
+
"sync_checkpoints": false,
|
| 53 |
+
"rollout_engine": "vllm",
|
| 54 |
+
"vllm_gpu": "2",
|
| 55 |
+
"vllm_port": 8377,
|
| 56 |
+
"vllm_refresh_every": 1,
|
| 57 |
+
"vllm_serve_bin": "vllm-plugin/.venv/bin/python",
|
| 58 |
+
"vllm_gpu_mem_util": 0.85,
|
| 59 |
+
"liger_loss": true,
|
| 60 |
+
"gold_mix_lambda": 0.5,
|
| 61 |
+
"gold_topk_targets": "outputs/teacher_trajectories/dolci_combined_top128",
|
| 62 |
+
"gold_loss": "ce",
|
| 63 |
+
"gold_mix_decay": 0.0,
|
| 64 |
+
"fast_teacher": true,
|
| 65 |
+
"reference_kl_beta": 0.05,
|
| 66 |
+
"drop_truncated_rollouts": false,
|
| 67 |
+
"vllm_max_model_len": null,
|
| 68 |
+
"vllm_refresh_mode": "reload",
|
| 69 |
+
"vllm_live_dir": null,
|
| 70 |
+
"resolved_kl_direction": "reverse"
|
| 71 |
+
}
|
healed/stableopd_cold_keep50/train_log.jsonl
ADDED
|
@@ -0,0 +1,127 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"step": 1, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6072952454558517, "tokens": 174656, "cumulative_loss_tokens": 174656, "grad_norm": 4.1875, "lr": 6.000000000000001e-07, "finish_rate": 0.98, "comp_len": 682.2, "dropped_truncated": 0, "gold_loss": 0.4598, "gold_lambda": 0.5, "rep_ratio": 2.51, "t_data_s": 0.0, "t_rollout_s": 53.2, "t_step_s": 136.4, "t_refresh_s": 0.3, "mem_gb": 10.54, "mem_gb_teacher": 20.6}
|
| 2 |
+
{"step": 1, "gsm8k_n": 256, "gsm8k_quick_chat": 0.58984375, "t_eval_s": 23.8}
|
| 3 |
+
{"step": 2, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5839993848137889, "tokens": 179858, "cumulative_loss_tokens": 354514, "grad_norm": 4.21875, "lr": 9e-07, "finish_rate": 0.988, "comp_len": 702.6, "dropped_truncated": 0, "gold_loss": 0.4407, "gold_lambda": 0.5, "rep_ratio": 2.417, "t_data_s": 0.0, "t_rollout_s": 54.2, "t_step_s": 128.6, "t_refresh_s": 0.3, "mem_gb": 10.21, "mem_gb_teacher": 20.55}
|
| 4 |
+
{"step": 3, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6450184482514114, "tokens": 187011, "cumulative_loss_tokens": 541525, "grad_norm": 3.859375, "lr": 1.2000000000000002e-06, "finish_rate": 0.949, "comp_len": 730.5, "dropped_truncated": 0, "gold_loss": 0.4914, "gold_lambda": 0.5, "rep_ratio": 2.484, "t_data_s": 0.0, "t_rollout_s": 55.7, "t_step_s": 133.8, "t_refresh_s": 0.3, "mem_gb": 10.43, "mem_gb_teacher": 20.63}
|
| 5 |
+
{"step": 4, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5364472807322356, "tokens": 163584, "cumulative_loss_tokens": 705109, "grad_norm": 3.734375, "lr": 1.5e-06, "finish_rate": 0.984, "comp_len": 639.0, "dropped_truncated": 0, "gold_loss": 0.4841, "gold_lambda": 0.5, "rep_ratio": 2.572, "t_data_s": 0.0, "t_rollout_s": 49.3, "t_step_s": 123.3, "t_refresh_s": 0.3, "mem_gb": 10.29, "mem_gb_teacher": 20.57}
|
| 6 |
+
{"step": 5, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6003048529689194, "tokens": 163316, "cumulative_loss_tokens": 868425, "grad_norm": 5.1875, "lr": 1.8e-06, "finish_rate": 0.996, "comp_len": 638.0, "dropped_truncated": 0, "gold_loss": 1.5716, "gold_lambda": 0.5, "rep_ratio": 2.618, "t_data_s": 0.0, "t_rollout_s": 49.3, "t_step_s": 120.6, "t_refresh_s": 0.3, "mem_gb": 10.14, "mem_gb_teacher": 20.53}
|
| 7 |
+
{"step": 6, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5325979884647087, "tokens": 166966, "cumulative_loss_tokens": 1035391, "grad_norm": 4.1875, "lr": 2.1e-06, "finish_rate": 0.992, "comp_len": 652.2, "dropped_truncated": 0, "gold_loss": 1.2164, "gold_lambda": 0.5, "rep_ratio": 2.318, "t_data_s": 0.0, "t_rollout_s": 53.1, "t_step_s": 125.4, "t_refresh_s": 0.3, "mem_gb": 10.29, "mem_gb_teacher": 20.58}
|
| 8 |
+
{"step": 7, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4701439193907984, "tokens": 185189, "cumulative_loss_tokens": 1220580, "grad_norm": 3.640625, "lr": 2.4000000000000003e-06, "finish_rate": 0.988, "comp_len": 723.4, "dropped_truncated": 0, "gold_loss": 1.5872, "gold_lambda": 0.5, "rep_ratio": 2.368, "t_data_s": 0.0, "t_rollout_s": 55.8, "t_step_s": 133.3, "t_refresh_s": 0.3, "mem_gb": 10.25, "mem_gb_teacher": 20.53}
|
| 9 |
+
{"step": 8, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5297812680536196, "tokens": 169129, "cumulative_loss_tokens": 1389709, "grad_norm": 4.15625, "lr": 2.7e-06, "finish_rate": 0.992, "comp_len": 660.7, "dropped_truncated": 0, "gold_loss": 1.6643, "gold_lambda": 0.5, "rep_ratio": 2.508, "t_data_s": 0.0, "t_rollout_s": 50.9, "t_step_s": 123.1, "t_refresh_s": 0.3, "mem_gb": 10.3, "mem_gb_teacher": 20.52}
|
| 10 |
+
{"step": 9, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6911732661994033, "tokens": 181954, "cumulative_loss_tokens": 1571663, "grad_norm": 4.90625, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 710.8, "dropped_truncated": 0, "gold_loss": 0.5111, "gold_lambda": 0.5, "rep_ratio": 2.326, "t_data_s": 0.0, "t_rollout_s": 54.8, "t_step_s": 130.7, "t_refresh_s": 0.3, "mem_gb": 10.58, "mem_gb_teacher": 20.6}
|
| 11 |
+
{"step": 10, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5809754839900768, "tokens": 178398, "cumulative_loss_tokens": 1750061, "grad_norm": 3.75, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 696.9, "dropped_truncated": 0, "gold_loss": 0.4964, "gold_lambda": 0.5, "rep_ratio": 2.747, "t_data_s": 0.0, "t_rollout_s": 54.1, "t_step_s": 129.5, "t_refresh_s": 0.3, "mem_gb": 10.88, "mem_gb_teacher": 20.71}
|
| 12 |
+
{"step": 11, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5429350640162725, "tokens": 164601, "cumulative_loss_tokens": 1914662, "grad_norm": 3.90625, "lr": 3e-06, "finish_rate": 0.996, "comp_len": 643.0, "dropped_truncated": 0, "gold_loss": 1.8407, "gold_lambda": 0.5, "rep_ratio": 2.687, "t_data_s": 0.0, "t_rollout_s": 50.0, "t_step_s": 120.8, "t_refresh_s": 0.3, "mem_gb": 10.33, "mem_gb_teacher": 20.52}
|
| 13 |
+
{"step": 12, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5013015017285535, "tokens": 171628, "cumulative_loss_tokens": 2086290, "grad_norm": 3.8125, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 670.4, "dropped_truncated": 0, "gold_loss": 1.669, "gold_lambda": 0.5, "rep_ratio": 2.506, "t_data_s": 0.0, "t_rollout_s": 51.4, "t_step_s": 125.0, "t_refresh_s": 0.3, "mem_gb": 10.3, "mem_gb_teacher": 20.59}
|
| 14 |
+
{"step": 13, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6719065286703264, "tokens": 174929, "cumulative_loss_tokens": 2261219, "grad_norm": 5.78125, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 683.3, "dropped_truncated": 0, "gold_loss": 1.749, "gold_lambda": 0.5, "rep_ratio": 2.195, "t_data_s": 0.0, "t_rollout_s": 52.1, "t_step_s": 125.2, "t_refresh_s": 0.3, "mem_gb": 10.33, "mem_gb_teacher": 20.59}
|
| 15 |
+
{"step": 14, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6088382444373247, "tokens": 163031, "cumulative_loss_tokens": 2424250, "grad_norm": 4.25, "lr": 3e-06, "finish_rate": 0.996, "comp_len": 636.8, "dropped_truncated": 0, "gold_loss": 1.7023, "gold_lambda": 0.5, "rep_ratio": 2.52, "t_data_s": 0.0, "t_rollout_s": 49.6, "t_step_s": 120.2, "t_refresh_s": 0.3, "mem_gb": 10.07, "mem_gb_teacher": 20.5}
|
| 16 |
+
{"step": 15, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6188558125826603, "tokens": 179647, "cumulative_loss_tokens": 2603897, "grad_norm": 4.5625, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 701.7, "dropped_truncated": 0, "gold_loss": 0.4678, "gold_lambda": 0.5, "rep_ratio": 2.481, "t_data_s": 0.0, "t_rollout_s": 52.2, "t_step_s": 125.6, "t_refresh_s": 0.3, "mem_gb": 10.38, "mem_gb_teacher": 20.54}
|
| 17 |
+
{"step": 16, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5876797942654096, "tokens": 175418, "cumulative_loss_tokens": 2779315, "grad_norm": 3.5625, "lr": 3e-06, "finish_rate": 0.961, "comp_len": 685.2, "dropped_truncated": 0, "gold_loss": 0.454, "gold_lambda": 0.5, "rep_ratio": 2.716, "t_data_s": 0.0, "t_rollout_s": 51.6, "t_step_s": 124.9, "t_refresh_s": 0.3, "mem_gb": 10.31, "mem_gb_teacher": 20.57}
|
| 18 |
+
{"step": 17, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7039823117514916, "tokens": 155950, "cumulative_loss_tokens": 2935265, "grad_norm": 6.0, "lr": 3e-06, "finish_rate": 1.0, "comp_len": 609.2, "dropped_truncated": 0, "gold_loss": 1.3804, "gold_lambda": 0.5, "rep_ratio": 2.378, "t_data_s": 0.0, "t_rollout_s": 48.0, "t_step_s": 116.4, "t_refresh_s": 0.3, "mem_gb": 10.28, "mem_gb_teacher": 20.58}
|
| 19 |
+
{"step": 18, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6408077914539468, "tokens": 175083, "cumulative_loss_tokens": 3110348, "grad_norm": 4.90625, "lr": 3e-06, "finish_rate": 1.0, "comp_len": 683.9, "dropped_truncated": 0, "gold_loss": 1.4034, "gold_lambda": 0.5, "rep_ratio": 2.388, "t_data_s": 0.0, "t_rollout_s": 54.3, "t_step_s": 128.9, "t_refresh_s": 0.3, "mem_gb": 10.5, "mem_gb_teacher": 20.65}
|
| 20 |
+
{"step": 19, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6064575240328233, "tokens": 174901, "cumulative_loss_tokens": 3285249, "grad_norm": 4.46875, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 683.2, "dropped_truncated": 0, "gold_loss": 1.6341, "gold_lambda": 0.5, "rep_ratio": 2.478, "t_data_s": 0.0, "t_rollout_s": 53.1, "t_step_s": 126.5, "t_refresh_s": 0.3, "mem_gb": 10.44, "mem_gb_teacher": 20.55}
|
| 21 |
+
{"step": 20, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4756010118007534, "tokens": 188227, "cumulative_loss_tokens": 3473476, "grad_norm": 3.140625, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 735.3, "dropped_truncated": 0, "gold_loss": 1.8114, "gold_lambda": 0.5, "rep_ratio": 2.635, "t_data_s": 0.0, "t_rollout_s": 55.7, "t_step_s": 134.4, "t_refresh_s": 0.3, "mem_gb": 10.63, "mem_gb_teacher": 20.64}
|
| 22 |
+
{"step": 20, "gsm8k_n": 256, "gsm8k_quick_chat": 0.58984375, "t_eval_s": 21.8}
|
| 23 |
+
{"step": 21, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6079329543972215, "tokens": 162079, "cumulative_loss_tokens": 3635555, "grad_norm": 3.734375, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 633.1, "dropped_truncated": 0, "gold_loss": 0.4071, "gold_lambda": 0.5, "rep_ratio": 2.638, "t_data_s": 0.0, "t_rollout_s": 49.2, "t_step_s": 121.3, "t_refresh_s": 0.3, "mem_gb": 10.41, "mem_gb_teacher": 20.55}
|
| 24 |
+
{"step": 22, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5545496602545528, "tokens": 187630, "cumulative_loss_tokens": 3823185, "grad_norm": 3.515625, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 732.9, "dropped_truncated": 0, "gold_loss": 0.4184, "gold_lambda": 0.5, "rep_ratio": 2.805, "t_data_s": 0.0, "t_rollout_s": 56.9, "t_step_s": 136.7, "t_refresh_s": 0.3, "mem_gb": 10.3, "mem_gb_teacher": 20.58}
|
| 25 |
+
{"step": 23, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6361210861404402, "tokens": 163175, "cumulative_loss_tokens": 3986360, "grad_norm": 5.0625, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 637.4, "dropped_truncated": 0, "gold_loss": 1.0212, "gold_lambda": 0.5, "rep_ratio": 2.42, "t_data_s": 0.0, "t_rollout_s": 48.9, "t_step_s": 120.4, "t_refresh_s": 0.3, "mem_gb": 10.14, "mem_gb_teacher": 20.53}
|
| 26 |
+
{"step": 24, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.599811350401479, "tokens": 163397, "cumulative_loss_tokens": 4149757, "grad_norm": 4.71875, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 638.3, "dropped_truncated": 0, "gold_loss": 1.0594, "gold_lambda": 0.5, "rep_ratio": 2.544, "t_data_s": 0.0, "t_rollout_s": 49.5, "t_step_s": 118.7, "t_refresh_s": 0.3, "mem_gb": 10.7, "mem_gb_teacher": 20.65}
|
| 27 |
+
{"step": 25, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5804433731037768, "tokens": 199477, "cumulative_loss_tokens": 4349234, "grad_norm": 4.4375, "lr": 3e-06, "finish_rate": 0.965, "comp_len": 779.2, "dropped_truncated": 0, "gold_loss": 1.9939, "gold_lambda": 0.5, "rep_ratio": 2.837, "t_data_s": 0.0, "t_rollout_s": 59.4, "t_step_s": 136.4, "t_refresh_s": 0.3, "mem_gb": 10.56, "mem_gb_teacher": 20.67}
|
| 28 |
+
{"step": 26, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.497977049129856, "tokens": 182212, "cumulative_loss_tokens": 4531446, "grad_norm": 4.1875, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 711.8, "dropped_truncated": 0, "gold_loss": 1.7395, "gold_lambda": 0.5, "rep_ratio": 2.455, "t_data_s": 0.0, "t_rollout_s": 55.3, "t_step_s": 129.3, "t_refresh_s": 0.3, "mem_gb": 10.38, "mem_gb_teacher": 20.59}
|
| 29 |
+
{"step": 27, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6490434516885443, "tokens": 178833, "cumulative_loss_tokens": 4710279, "grad_norm": 5.4375, "lr": 3e-06, "finish_rate": 0.965, "comp_len": 698.6, "dropped_truncated": 0, "gold_loss": 1.246, "gold_lambda": 0.5, "rep_ratio": 2.638, "t_data_s": 0.0, "t_rollout_s": 53.8, "t_step_s": 129.3, "t_refresh_s": 0.3, "mem_gb": 10.39, "mem_gb_teacher": 20.62}
|
| 30 |
+
{"step": 28, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5880148602511409, "tokens": 179375, "cumulative_loss_tokens": 4889654, "grad_norm": 4.6875, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 700.7, "dropped_truncated": 0, "gold_loss": 1.322, "gold_lambda": 0.5, "rep_ratio": 2.433, "t_data_s": 0.0, "t_rollout_s": 53.2, "t_step_s": 128.5, "t_refresh_s": 0.3, "mem_gb": 10.25, "mem_gb_teacher": 20.54}
|
| 31 |
+
{"step": 29, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5745216805737519, "tokens": 166517, "cumulative_loss_tokens": 5056171, "grad_norm": 3.84375, "lr": 3e-06, "finish_rate": 0.969, "comp_len": 650.5, "dropped_truncated": 0, "gold_loss": 0.5016, "gold_lambda": 0.5, "rep_ratio": 2.456, "t_data_s": 0.0, "t_rollout_s": 51.9, "t_step_s": 126.7, "t_refresh_s": 0.3, "mem_gb": 10.38, "mem_gb_teacher": 20.6}
|
| 32 |
+
{"step": 30, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5298360355312054, "tokens": 183134, "cumulative_loss_tokens": 5239305, "grad_norm": 2.984375, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 715.4, "dropped_truncated": 0, "gold_loss": 0.4799, "gold_lambda": 0.5, "rep_ratio": 2.54, "t_data_s": 0.0, "t_rollout_s": 54.7, "t_step_s": 129.5, "t_refresh_s": 0.3, "mem_gb": 10.61, "mem_gb_teacher": 20.62}
|
| 33 |
+
{"step": 31, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5756465768027541, "tokens": 196051, "cumulative_loss_tokens": 5435356, "grad_norm": 4.15625, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 765.8, "dropped_truncated": 0, "gold_loss": 1.5787, "gold_lambda": 0.5, "rep_ratio": 2.534, "t_data_s": 0.0, "t_rollout_s": 59.3, "t_step_s": 139.0, "t_refresh_s": 0.3, "mem_gb": 10.26, "mem_gb_teacher": 20.6}
|
| 34 |
+
{"step": 32, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5410583641016259, "tokens": 189330, "cumulative_loss_tokens": 5624686, "grad_norm": 3.921875, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 739.6, "dropped_truncated": 0, "gold_loss": 1.3354, "gold_lambda": 0.5, "rep_ratio": 2.64, "t_data_s": 0.0, "t_rollout_s": 57.8, "t_step_s": 137.7, "t_refresh_s": 0.3, "mem_gb": 10.64, "mem_gb_teacher": 20.7}
|
| 35 |
+
{"step": 33, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6491864148698413, "tokens": 175189, "cumulative_loss_tokens": 5799875, "grad_norm": 5.125, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 684.3, "dropped_truncated": 0, "gold_loss": 1.7111, "gold_lambda": 0.5, "rep_ratio": 2.595, "t_data_s": 0.0, "t_rollout_s": 53.3, "t_step_s": 130.5, "t_refresh_s": 0.3, "mem_gb": 10.55, "mem_gb_teacher": 20.66}
|
| 36 |
+
{"step": 34, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5394474143388251, "tokens": 172904, "cumulative_loss_tokens": 5972779, "grad_norm": 3.34375, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 675.4, "dropped_truncated": 0, "gold_loss": 2.158, "gold_lambda": 0.5, "rep_ratio": 2.501, "t_data_s": 0.0, "t_rollout_s": 53.3, "t_step_s": 129.6, "t_refresh_s": 0.3, "mem_gb": 10.4, "mem_gb_teacher": 20.6}
|
| 37 |
+
{"step": 35, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5962679599354465, "tokens": 171947, "cumulative_loss_tokens": 6144726, "grad_norm": 3.90625, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 671.7, "dropped_truncated": 0, "gold_loss": 0.3504, "gold_lambda": 0.5, "rep_ratio": 2.467, "t_data_s": 0.0, "t_rollout_s": 52.1, "t_step_s": 126.9, "t_refresh_s": 0.3, "mem_gb": 10.35, "mem_gb_teacher": 20.56}
|
| 38 |
+
{"step": 36, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6593742652597039, "tokens": 173200, "cumulative_loss_tokens": 6317926, "grad_norm": 3.96875, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 676.6, "dropped_truncated": 0, "gold_loss": 0.4003, "gold_lambda": 0.5, "rep_ratio": 2.488, "t_data_s": 0.0, "t_rollout_s": 52.7, "t_step_s": 125.6, "t_refresh_s": 0.3, "mem_gb": 10.46, "mem_gb_teacher": 20.57}
|
| 39 |
+
{"step": 37, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6162966957118422, "tokens": 172175, "cumulative_loss_tokens": 6490101, "grad_norm": 4.0625, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 672.6, "dropped_truncated": 0, "gold_loss": 1.4778, "gold_lambda": 0.5, "rep_ratio": 2.377, "t_data_s": 0.0, "t_rollout_s": 52.0, "t_step_s": 124.7, "t_refresh_s": 0.3, "mem_gb": 10.27, "mem_gb_teacher": 20.51}
|
| 40 |
+
{"step": 38, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.48704485445759893, "tokens": 192333, "cumulative_loss_tokens": 6682434, "grad_norm": 2.734375, "lr": 3e-06, "finish_rate": 0.969, "comp_len": 751.3, "dropped_truncated": 0, "gold_loss": 1.4379, "gold_lambda": 0.5, "rep_ratio": 2.396, "t_data_s": 0.0, "t_rollout_s": 55.0, "t_step_s": 131.6, "t_refresh_s": 0.3, "mem_gb": 10.28, "mem_gb_teacher": 20.51}
|
| 41 |
+
{"step": 39, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4630184856648492, "tokens": 189244, "cumulative_loss_tokens": 6871678, "grad_norm": 2.25, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 739.2, "dropped_truncated": 0, "gold_loss": 0.3596, "gold_lambda": 0.5, "rep_ratio": 2.667, "t_data_s": 0.0, "t_rollout_s": 56.9, "t_step_s": 136.1, "t_refresh_s": 0.3, "mem_gb": 10.56, "mem_gb_teacher": 20.61}
|
| 42 |
+
{"step": 40, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6224862955720186, "tokens": 170115, "cumulative_loss_tokens": 7041793, "grad_norm": 3.5625, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 664.5, "dropped_truncated": 0, "gold_loss": 0.3538, "gold_lambda": 0.5, "rep_ratio": 2.749, "t_data_s": 0.0, "t_rollout_s": 49.8, "t_step_s": 121.0, "t_refresh_s": 0.3, "mem_gb": 10.28, "mem_gb_teacher": 20.51}
|
| 43 |
+
{"step": 40, "gsm8k_n": 256, "gsm8k_quick_chat": 0.6015625, "t_eval_s": 21.7}
|
| 44 |
+
{"step": 41, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5495724419499659, "tokens": 172583, "cumulative_loss_tokens": 7214376, "grad_norm": 3.984375, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 674.2, "dropped_truncated": 0, "gold_loss": 1.4043, "gold_lambda": 0.5, "rep_ratio": 2.542, "t_data_s": 0.0, "t_rollout_s": 49.7, "t_step_s": 120.0, "t_refresh_s": 0.3, "mem_gb": 10.25, "mem_gb_teacher": 20.53}
|
| 45 |
+
{"step": 42, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5940980768232321, "tokens": 168268, "cumulative_loss_tokens": 7382644, "grad_norm": 3.8125, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 657.3, "dropped_truncated": 0, "gold_loss": 1.3572, "gold_lambda": 0.5, "rep_ratio": 2.601, "t_data_s": 0.0, "t_rollout_s": 49.7, "t_step_s": 118.2, "t_refresh_s": 0.3, "mem_gb": 10.33, "mem_gb_teacher": 20.52}
|
| 46 |
+
{"step": 43, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5012217065778929, "tokens": 170014, "cumulative_loss_tokens": 7552658, "grad_norm": 2.96875, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 664.1, "dropped_truncated": 0, "gold_loss": 1.6534, "gold_lambda": 0.5, "rep_ratio": 2.477, "t_data_s": 0.0, "t_rollout_s": 51.4, "t_step_s": 125.4, "t_refresh_s": 0.3, "mem_gb": 10.31, "mem_gb_teacher": 20.56}
|
| 47 |
+
{"step": 44, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6404573169656133, "tokens": 163289, "cumulative_loss_tokens": 7715947, "grad_norm": 4.03125, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 637.8, "dropped_truncated": 0, "gold_loss": 1.857, "gold_lambda": 0.5, "rep_ratio": 2.391, "t_data_s": 0.0, "t_rollout_s": 48.9, "t_step_s": 117.9, "t_refresh_s": 0.3, "mem_gb": 10.4, "mem_gb_teacher": 20.56}
|
| 48 |
+
{"step": 45, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5974512944189411, "tokens": 179043, "cumulative_loss_tokens": 7894990, "grad_norm": 4.8125, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 699.4, "dropped_truncated": 0, "gold_loss": 1.4153, "gold_lambda": 0.5, "rep_ratio": 2.418, "t_data_s": 0.0, "t_rollout_s": 52.4, "t_step_s": 128.6, "t_refresh_s": 0.3, "mem_gb": 10.44, "mem_gb_teacher": 20.59}
|
| 49 |
+
{"step": 46, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.46899697193267115, "tokens": 177867, "cumulative_loss_tokens": 8072857, "grad_norm": 3.265625, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 694.8, "dropped_truncated": 0, "gold_loss": 1.4124, "gold_lambda": 0.5, "rep_ratio": 2.706, "t_data_s": 0.0, "t_rollout_s": 52.5, "t_step_s": 125.6, "t_refresh_s": 0.3, "mem_gb": 10.29, "mem_gb_teacher": 20.54}
|
| 50 |
+
{"step": 47, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.585951602252788, "tokens": 185522, "cumulative_loss_tokens": 8258379, "grad_norm": 3.125, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 724.7, "dropped_truncated": 0, "gold_loss": 0.5289, "gold_lambda": 0.5, "rep_ratio": 2.489, "t_data_s": 0.0, "t_rollout_s": 56.9, "t_step_s": 135.7, "t_refresh_s": 0.3, "mem_gb": 10.73, "mem_gb_teacher": 20.74}
|
| 51 |
+
{"step": 48, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4633422546430412, "tokens": 169129, "cumulative_loss_tokens": 8427508, "grad_norm": 2.5625, "lr": 3e-06, "finish_rate": 0.996, "comp_len": 660.7, "dropped_truncated": 0, "gold_loss": 0.5208, "gold_lambda": 0.5, "rep_ratio": 2.541, "t_data_s": 0.0, "t_rollout_s": 51.2, "t_step_s": 124.1, "t_refresh_s": 0.3, "mem_gb": 10.04, "mem_gb_teacher": 20.5}
|
| 52 |
+
{"step": 49, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5922208327087102, "tokens": 171625, "cumulative_loss_tokens": 8599133, "grad_norm": 3.359375, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 670.4, "dropped_truncated": 0, "gold_loss": 1.7226, "gold_lambda": 0.5, "rep_ratio": 2.5, "t_data_s": 0.0, "t_rollout_s": 51.5, "t_step_s": 123.3, "t_refresh_s": 0.3, "mem_gb": 10.18, "mem_gb_teacher": 20.54}
|
| 53 |
+
{"step": 50, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5572641377977327, "tokens": 161003, "cumulative_loss_tokens": 8760136, "grad_norm": 3.609375, "lr": 3e-06, "finish_rate": 1.0, "comp_len": 628.9, "dropped_truncated": 0, "gold_loss": 1.4478, "gold_lambda": 0.5, "rep_ratio": 2.39, "t_data_s": 0.0, "t_rollout_s": 49.2, "t_step_s": 120.5, "t_refresh_s": 0.3, "mem_gb": 10.32, "mem_gb_teacher": 20.59}
|
| 54 |
+
{"step": 51, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6542427273296131, "tokens": 170048, "cumulative_loss_tokens": 8930184, "grad_norm": 4.15625, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 664.2, "dropped_truncated": 0, "gold_loss": 1.4075, "gold_lambda": 0.5, "rep_ratio": 2.532, "t_data_s": 0.0, "t_rollout_s": 50.4, "t_step_s": 120.9, "t_refresh_s": 0.3, "mem_gb": 10.29, "mem_gb_teacher": 20.58}
|
| 55 |
+
{"step": 52, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6006482859702559, "tokens": 157639, "cumulative_loss_tokens": 9087823, "grad_norm": 3.609375, "lr": 3e-06, "finish_rate": 0.996, "comp_len": 615.8, "dropped_truncated": 0, "gold_loss": 1.0789, "gold_lambda": 0.5, "rep_ratio": 2.607, "t_data_s": 0.0, "t_rollout_s": 48.7, "t_step_s": 119.3, "t_refresh_s": 0.3, "mem_gb": 10.26, "mem_gb_teacher": 20.56}
|
| 56 |
+
{"step": 53, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5860928311645158, "tokens": 179602, "cumulative_loss_tokens": 9267425, "grad_norm": 3.78125, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 701.6, "dropped_truncated": 0, "gold_loss": 1.1582, "gold_lambda": 0.5, "rep_ratio": 3.591, "t_data_s": 0.0, "t_rollout_s": 54.6, "t_step_s": 130.6, "t_refresh_s": 0.3, "mem_gb": 10.49, "mem_gb_teacher": 20.65}
|
| 57 |
+
{"step": 54, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5360958047571189, "tokens": 189980, "cumulative_loss_tokens": 9457405, "grad_norm": 3.625, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 742.1, "dropped_truncated": 0, "gold_loss": 1.3728, "gold_lambda": 0.5, "rep_ratio": 2.641, "t_data_s": 0.0, "t_rollout_s": 56.6, "t_step_s": 134.8, "t_refresh_s": 0.3, "mem_gb": 10.61, "mem_gb_teacher": 20.63}
|
| 58 |
+
{"step": 55, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.615571619503994, "tokens": 175860, "cumulative_loss_tokens": 9633265, "grad_norm": 4.0625, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 687.0, "dropped_truncated": 0, "gold_loss": 2.1235, "gold_lambda": 0.5, "rep_ratio": 2.376, "t_data_s": 0.0, "t_rollout_s": 53.9, "t_step_s": 127.7, "t_refresh_s": 0.3, "mem_gb": 10.55, "mem_gb_teacher": 20.67}
|
| 59 |
+
{"step": 56, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6498361152531915, "tokens": 179539, "cumulative_loss_tokens": 9812804, "grad_norm": 4.5, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 701.3, "dropped_truncated": 0, "gold_loss": 1.4776, "gold_lambda": 0.5, "rep_ratio": 2.319, "t_data_s": 0.0, "t_rollout_s": 54.2, "t_step_s": 130.2, "t_refresh_s": 0.3, "mem_gb": 10.43, "mem_gb_teacher": 20.55}
|
| 60 |
+
{"step": 57, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6616428990422922, "tokens": 158553, "cumulative_loss_tokens": 9971357, "grad_norm": 4.65625, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 619.3, "dropped_truncated": 0, "gold_loss": 1.2563, "gold_lambda": 0.5, "rep_ratio": 2.639, "t_data_s": 0.0, "t_rollout_s": 48.6, "t_step_s": 120.1, "t_refresh_s": 0.3, "mem_gb": 10.36, "mem_gb_teacher": 20.61}
|
| 61 |
+
{"step": 58, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4762311771429326, "tokens": 170983, "cumulative_loss_tokens": 10142340, "grad_norm": 2.65625, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 667.9, "dropped_truncated": 0, "gold_loss": 1.5603, "gold_lambda": 0.5, "rep_ratio": 2.476, "t_data_s": 0.0, "t_rollout_s": 51.9, "t_step_s": 124.1, "t_refresh_s": 0.3, "mem_gb": 10.27, "mem_gb_teacher": 20.57}
|
| 62 |
+
{"step": 59, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.48560476492901544, "tokens": 188980, "cumulative_loss_tokens": 10331320, "grad_norm": 3.484375, "lr": 3e-06, "finish_rate": 0.965, "comp_len": 738.2, "dropped_truncated": 0, "gold_loss": 1.3256, "gold_lambda": 0.5, "rep_ratio": 2.832, "t_data_s": 0.0, "t_rollout_s": 56.1, "t_step_s": 132.8, "t_refresh_s": 0.3, "mem_gb": 10.44, "mem_gb_teacher": 20.56}
|
| 63 |
+
{"step": 60, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.547719864326989, "tokens": 163676, "cumulative_loss_tokens": 10494996, "grad_norm": 3.84375, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 639.4, "dropped_truncated": 0, "gold_loss": 1.668, "gold_lambda": 0.5, "rep_ratio": 2.554, "t_data_s": 0.0, "t_rollout_s": 49.9, "t_step_s": 121.7, "t_refresh_s": 0.3, "mem_gb": 10.44, "mem_gb_teacher": 20.56}
|
| 64 |
+
{"step": 60, "gsm8k_n": 256, "gsm8k_quick_chat": 0.59765625, "t_eval_s": 21.7}
|
| 65 |
+
{"step": 61, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5137778637439316, "tokens": 189784, "cumulative_loss_tokens": 10684780, "grad_norm": 3.21875, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 741.3, "dropped_truncated": 0, "gold_loss": 0.4965, "gold_lambda": 0.5, "rep_ratio": 2.478, "t_data_s": 0.0, "t_rollout_s": 54.4, "t_step_s": 130.2, "t_refresh_s": 0.3, "mem_gb": 10.15, "mem_gb_teacher": 20.53}
|
| 66 |
+
{"step": 62, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5286864652203259, "tokens": 191147, "cumulative_loss_tokens": 10875927, "grad_norm": 3.109375, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 746.7, "dropped_truncated": 0, "gold_loss": 0.5175, "gold_lambda": 0.5, "rep_ratio": 2.456, "t_data_s": 0.0, "t_rollout_s": 56.1, "t_step_s": 143.9, "t_refresh_s": 0.3, "mem_gb": 10.36, "mem_gb_teacher": 20.59}
|
| 67 |
+
{"step": 63, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.49351737753958147, "tokens": 179397, "cumulative_loss_tokens": 11055324, "grad_norm": 2.640625, "lr": 3e-06, "finish_rate": 0.996, "comp_len": 700.8, "dropped_truncated": 0, "gold_loss": 0.5205, "gold_lambda": 0.5, "rep_ratio": 2.602, "t_data_s": 0.0, "t_rollout_s": 53.1, "t_step_s": 128.7, "t_refresh_s": 0.3, "mem_gb": 10.21, "mem_gb_teacher": 20.54}
|
| 68 |
+
{"step": 64, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6464224095801929, "tokens": 181006, "cumulative_loss_tokens": 11236330, "grad_norm": 3.078125, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 707.1, "dropped_truncated": 0, "gold_loss": 0.4944, "gold_lambda": 0.5, "rep_ratio": 2.528, "t_data_s": 0.0, "t_rollout_s": 55.7, "t_step_s": 133.9, "t_refresh_s": 0.3, "mem_gb": 10.28, "mem_gb_teacher": 20.58}
|
| 69 |
+
{"step": 65, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6314773864732354, "tokens": 157991, "cumulative_loss_tokens": 11394321, "grad_norm": 3.796875, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 617.2, "dropped_truncated": 0, "gold_loss": 1.2006, "gold_lambda": 0.5, "rep_ratio": 2.6, "t_data_s": 0.0, "t_rollout_s": 48.8, "t_step_s": 121.3, "t_refresh_s": 0.3, "mem_gb": 10.41, "mem_gb_teacher": 20.56}
|
| 70 |
+
{"step": 66, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5152227263231336, "tokens": 170045, "cumulative_loss_tokens": 11564366, "grad_norm": 3.28125, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 664.2, "dropped_truncated": 0, "gold_loss": 1.6976, "gold_lambda": 0.5, "rep_ratio": 2.466, "t_data_s": 0.0, "t_rollout_s": 51.0, "t_step_s": 120.6, "t_refresh_s": 0.3, "mem_gb": 10.2, "mem_gb_teacher": 20.55}
|
| 71 |
+
{"step": 67, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.48801385307029, "tokens": 177712, "cumulative_loss_tokens": 11742078, "grad_norm": 3.640625, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 694.2, "dropped_truncated": 0, "gold_loss": 1.3421, "gold_lambda": 0.5, "rep_ratio": 2.556, "t_data_s": 0.0, "t_rollout_s": 51.6, "t_step_s": 121.4, "t_refresh_s": 0.3, "mem_gb": 10.38, "mem_gb_teacher": 20.6}
|
| 72 |
+
{"step": 68, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5286305613993042, "tokens": 173846, "cumulative_loss_tokens": 11915924, "grad_norm": 3.59375, "lr": 3e-06, "finish_rate": 0.996, "comp_len": 679.1, "dropped_truncated": 0, "gold_loss": 1.2664, "gold_lambda": 0.5, "rep_ratio": 2.648, "t_data_s": 0.0, "t_rollout_s": 51.9, "t_step_s": 123.0, "t_refresh_s": 0.3, "mem_gb": 10.09, "mem_gb_teacher": 20.51}
|
| 73 |
+
{"step": 69, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4648810376214836, "tokens": 179441, "cumulative_loss_tokens": 12095365, "grad_norm": 3.34375, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 700.9, "dropped_truncated": 0, "gold_loss": 1.7655, "gold_lambda": 0.5, "rep_ratio": 2.409, "t_data_s": 0.0, "t_rollout_s": 53.0, "t_step_s": 128.6, "t_refresh_s": 0.3, "mem_gb": 10.45, "mem_gb_teacher": 20.56}
|
| 74 |
+
{"step": 70, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5794084474774793, "tokens": 166734, "cumulative_loss_tokens": 12262099, "grad_norm": 4.3125, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 651.3, "dropped_truncated": 0, "gold_loss": 1.6365, "gold_lambda": 0.5, "rep_ratio": 2.309, "t_data_s": 0.0, "t_rollout_s": 50.4, "t_step_s": 123.4, "t_refresh_s": 0.3, "mem_gb": 10.27, "mem_gb_teacher": 20.52}
|
| 75 |
+
{"step": 71, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6559641185079966, "tokens": 150911, "cumulative_loss_tokens": 12413010, "grad_norm": 4.40625, "lr": 3e-06, "finish_rate": 0.996, "comp_len": 589.5, "dropped_truncated": 0, "gold_loss": 1.3251, "gold_lambda": 0.5, "rep_ratio": 2.7, "t_data_s": 0.0, "t_rollout_s": 46.6, "t_step_s": 114.2, "t_refresh_s": 0.3, "mem_gb": 10.38, "mem_gb_teacher": 20.55}
|
| 76 |
+
{"step": 72, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5143795466856095, "tokens": 189215, "cumulative_loss_tokens": 12602225, "grad_norm": 2.9375, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 739.1, "dropped_truncated": 0, "gold_loss": 1.6625, "gold_lambda": 0.5, "rep_ratio": 2.592, "t_data_s": 0.0, "t_rollout_s": 56.6, "t_step_s": 133.5, "t_refresh_s": 0.3, "mem_gb": 10.62, "mem_gb_teacher": 20.63}
|
| 77 |
+
{"step": 73, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5553455696392867, "tokens": 164849, "cumulative_loss_tokens": 12767074, "grad_norm": 2.9375, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 643.9, "dropped_truncated": 0, "gold_loss": 0.4704, "gold_lambda": 0.5, "rep_ratio": 2.381, "t_data_s": 0.0, "t_rollout_s": 50.7, "t_step_s": 124.8, "t_refresh_s": 0.3, "mem_gb": 10.31, "mem_gb_teacher": 20.58}
|
| 78 |
+
{"step": 74, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5939572483894898, "tokens": 175631, "cumulative_loss_tokens": 12942705, "grad_norm": 2.921875, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 686.1, "dropped_truncated": 0, "gold_loss": 0.4801, "gold_lambda": 0.5, "rep_ratio": 2.501, "t_data_s": 0.0, "t_rollout_s": 52.7, "t_step_s": 129.1, "t_refresh_s": 0.3, "mem_gb": 10.22, "mem_gb_teacher": 20.56}
|
| 79 |
+
{"step": 75, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5642269966953466, "tokens": 169179, "cumulative_loss_tokens": 13111884, "grad_norm": 4.15625, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 660.9, "dropped_truncated": 0, "gold_loss": 1.3203, "gold_lambda": 0.5, "rep_ratio": 2.467, "t_data_s": 0.0, "t_rollout_s": 50.7, "t_step_s": 121.3, "t_refresh_s": 0.3, "mem_gb": 10.3, "mem_gb_teacher": 20.58}
|
| 80 |
+
{"step": 76, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4545402607095275, "tokens": 189344, "cumulative_loss_tokens": 13301228, "grad_norm": 2.84375, "lr": 3e-06, "finish_rate": 0.965, "comp_len": 739.6, "dropped_truncated": 0, "gold_loss": 1.2242, "gold_lambda": 0.5, "rep_ratio": 2.578, "t_data_s": 0.0, "t_rollout_s": 55.3, "t_step_s": 130.1, "t_refresh_s": 0.3, "mem_gb": 10.53, "mem_gb_teacher": 20.59}
|
| 81 |
+
{"step": 77, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5516563839982034, "tokens": 181368, "cumulative_loss_tokens": 13482596, "grad_norm": 3.90625, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 708.5, "dropped_truncated": 0, "gold_loss": 1.2454, "gold_lambda": 0.5, "rep_ratio": 2.646, "t_data_s": 0.0, "t_rollout_s": 54.3, "t_step_s": 128.8, "t_refresh_s": 0.3, "mem_gb": 10.25, "mem_gb_teacher": 20.53}
|
| 82 |
+
{"step": 78, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6321706025450323, "tokens": 159314, "cumulative_loss_tokens": 13641910, "grad_norm": 4.1875, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 622.3, "dropped_truncated": 0, "gold_loss": 1.3701, "gold_lambda": 0.5, "rep_ratio": 2.529, "t_data_s": 0.0, "t_rollout_s": 49.8, "t_step_s": 133.8, "t_refresh_s": 0.3, "mem_gb": 10.27, "mem_gb_teacher": 20.57}
|
| 83 |
+
{"step": 79, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.49796059788262775, "tokens": 186037, "cumulative_loss_tokens": 13827947, "grad_norm": 3.4375, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 726.7, "dropped_truncated": 0, "gold_loss": 1.6689, "gold_lambda": 0.5, "rep_ratio": 2.565, "t_data_s": 0.0, "t_rollout_s": 54.7, "t_step_s": 129.3, "t_refresh_s": 0.3, "mem_gb": 10.21, "mem_gb_teacher": 20.55}
|
| 84 |
+
{"step": 80, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5557815555890879, "tokens": 166524, "cumulative_loss_tokens": 13994471, "grad_norm": 3.5625, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 650.5, "dropped_truncated": 0, "gold_loss": 1.6658, "gold_lambda": 0.5, "rep_ratio": 2.557, "t_data_s": 0.0, "t_rollout_s": 50.0, "t_step_s": 119.7, "t_refresh_s": 0.3, "mem_gb": 10.3, "mem_gb_teacher": 20.58}
|
| 85 |
+
{"step": 80, "gsm8k_n": 256, "gsm8k_quick_chat": 0.58984375, "t_eval_s": 12.4}
|
| 86 |
+
{"step": 81, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4678963162265957, "tokens": 186543, "cumulative_loss_tokens": 14181014, "grad_norm": 3.65625, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 728.7, "dropped_truncated": 0, "gold_loss": 1.56, "gold_lambda": 0.5, "rep_ratio": 2.713, "t_data_s": 0.0, "t_rollout_s": 56.2, "t_step_s": 133.1, "t_refresh_s": 0.3, "mem_gb": 10.22, "mem_gb_teacher": 20.52}
|
| 87 |
+
{"step": 82, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4774133586735581, "tokens": 189130, "cumulative_loss_tokens": 14370144, "grad_norm": 3.28125, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 738.8, "dropped_truncated": 0, "gold_loss": 1.7252, "gold_lambda": 0.5, "rep_ratio": 2.622, "t_data_s": 0.0, "t_rollout_s": 57.2, "t_step_s": 135.0, "t_refresh_s": 0.3, "mem_gb": 10.5, "mem_gb_teacher": 20.57}
|
| 88 |
+
{"step": 83, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5720705497632231, "tokens": 180727, "cumulative_loss_tokens": 14550871, "grad_norm": 4.53125, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 706.0, "dropped_truncated": 0, "gold_loss": 1.4518, "gold_lambda": 0.5, "rep_ratio": 2.381, "t_data_s": 0.0, "t_rollout_s": 55.1, "t_step_s": 130.5, "t_refresh_s": 0.3, "mem_gb": 10.22, "mem_gb_teacher": 20.55}
|
| 89 |
+
{"step": 84, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5261146624115828, "tokens": 189634, "cumulative_loss_tokens": 14740505, "grad_norm": 3.3125, "lr": 3e-06, "finish_rate": 0.957, "comp_len": 740.8, "dropped_truncated": 0, "gold_loss": 1.0758, "gold_lambda": 0.5, "rep_ratio": 2.206, "t_data_s": 0.0, "t_rollout_s": 56.5, "t_step_s": 131.5, "t_refresh_s": 0.3, "mem_gb": 10.45, "mem_gb_teacher": 20.6}
|
| 90 |
+
{"step": 85, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4498314615467012, "tokens": 188781, "cumulative_loss_tokens": 14929286, "grad_norm": 2.515625, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 737.4, "dropped_truncated": 0, "gold_loss": 1.2566, "gold_lambda": 0.5, "rep_ratio": 2.759, "t_data_s": 0.0, "t_rollout_s": 56.6, "t_step_s": 131.4, "t_refresh_s": 0.3, "mem_gb": 10.52, "mem_gb_teacher": 20.59}
|
| 91 |
+
{"step": 86, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5492913809256452, "tokens": 183231, "cumulative_loss_tokens": 15112517, "grad_norm": 3.125, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 715.7, "dropped_truncated": 0, "gold_loss": 1.1662, "gold_lambda": 0.5, "rep_ratio": 2.681, "t_data_s": 0.0, "t_rollout_s": 55.9, "t_step_s": 133.4, "t_refresh_s": 0.3, "mem_gb": 10.44, "mem_gb_teacher": 20.61}
|
| 92 |
+
{"step": 87, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5432304912490185, "tokens": 169663, "cumulative_loss_tokens": 15282180, "grad_norm": 3.859375, "lr": 3e-06, "finish_rate": 0.957, "comp_len": 662.7, "dropped_truncated": 0, "gold_loss": 1.4905, "gold_lambda": 0.5, "rep_ratio": 2.585, "t_data_s": 0.0, "t_rollout_s": 51.0, "t_step_s": 123.1, "t_refresh_s": 0.3, "mem_gb": 10.43, "mem_gb_teacher": 20.56}
|
| 93 |
+
{"step": 88, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4662917898026059, "tokens": 177987, "cumulative_loss_tokens": 15460167, "grad_norm": 3.28125, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 695.3, "dropped_truncated": 0, "gold_loss": 1.4101, "gold_lambda": 0.5, "rep_ratio": 2.606, "t_data_s": 0.0, "t_rollout_s": 52.8, "t_step_s": 125.2, "t_refresh_s": 0.3, "mem_gb": 10.36, "mem_gb_teacher": 20.6}
|
| 94 |
+
{"step": 89, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5261181109123355, "tokens": 197203, "cumulative_loss_tokens": 15657370, "grad_norm": 2.4375, "lr": 3e-06, "finish_rate": 0.961, "comp_len": 770.3, "dropped_truncated": 0, "gold_loss": 1.6196, "gold_lambda": 0.5, "rep_ratio": 2.457, "t_data_s": 0.0, "t_rollout_s": 57.9, "t_step_s": 136.5, "t_refresh_s": 0.3, "mem_gb": 10.3, "mem_gb_teacher": 20.52}
|
| 95 |
+
{"step": 90, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.572142635497917, "tokens": 192126, "cumulative_loss_tokens": 15849496, "grad_norm": 2.984375, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 750.5, "dropped_truncated": 0, "gold_loss": 1.6059, "gold_lambda": 0.5, "rep_ratio": 2.663, "t_data_s": 0.0, "t_rollout_s": 57.2, "t_step_s": 133.7, "t_refresh_s": 0.3, "mem_gb": 10.41, "mem_gb_teacher": 20.6}
|
| 96 |
+
{"step": 91, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.48959366036140506, "tokens": 198857, "cumulative_loss_tokens": 16048353, "grad_norm": 2.265625, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 776.8, "dropped_truncated": 0, "gold_loss": 0.5142, "gold_lambda": 0.5, "rep_ratio": 2.625, "t_data_s": 0.0, "t_rollout_s": 58.4, "t_step_s": 138.1, "t_refresh_s": 0.3, "mem_gb": 10.21, "mem_gb_teacher": 20.55}
|
| 97 |
+
{"step": 92, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5824183216032196, "tokens": 170037, "cumulative_loss_tokens": 16218390, "grad_norm": 3.296875, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 664.2, "dropped_truncated": 0, "gold_loss": 0.5594, "gold_lambda": 0.5, "rep_ratio": 2.578, "t_data_s": 0.0, "t_rollout_s": 50.2, "t_step_s": 121.8, "t_refresh_s": 0.3, "mem_gb": 10.4, "mem_gb_teacher": 20.54}
|
| 98 |
+
{"step": 93, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.45406014468650424, "tokens": 176539, "cumulative_loss_tokens": 16394929, "grad_norm": 2.78125, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 689.6, "dropped_truncated": 0, "gold_loss": 2.1234, "gold_lambda": 0.5, "rep_ratio": 2.535, "t_data_s": 0.0, "t_rollout_s": 53.0, "t_step_s": 128.3, "t_refresh_s": 0.3, "mem_gb": 10.26, "mem_gb_teacher": 20.54}
|
| 99 |
+
{"step": 94, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5435228508842771, "tokens": 183952, "cumulative_loss_tokens": 16578881, "grad_norm": 3.21875, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 718.6, "dropped_truncated": 0, "gold_loss": 1.8801, "gold_lambda": 0.5, "rep_ratio": 2.512, "t_data_s": 0.0, "t_rollout_s": 55.1, "t_step_s": 131.8, "t_refresh_s": 0.3, "mem_gb": 10.51, "mem_gb_teacher": 20.59}
|
| 100 |
+
{"step": 95, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5978032321878001, "tokens": 173404, "cumulative_loss_tokens": 16752285, "grad_norm": 3.03125, "lr": 3e-06, "finish_rate": 0.965, "comp_len": 677.4, "dropped_truncated": 0, "gold_loss": 1.7875, "gold_lambda": 0.5, "rep_ratio": 2.421, "t_data_s": 0.0, "t_rollout_s": 51.9, "t_step_s": 123.9, "t_refresh_s": 0.3, "mem_gb": 10.39, "mem_gb_teacher": 20.54}
|
| 101 |
+
{"step": 96, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5002416987161353, "tokens": 172230, "cumulative_loss_tokens": 16924515, "grad_norm": 3.125, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 672.8, "dropped_truncated": 0, "gold_loss": 1.5791, "gold_lambda": 0.5, "rep_ratio": 2.49, "t_data_s": 0.0, "t_rollout_s": 51.8, "t_step_s": 124.7, "t_refresh_s": 0.3, "mem_gb": 10.57, "mem_gb_teacher": 20.61}
|
| 102 |
+
{"step": 97, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4756894312193572, "tokens": 190312, "cumulative_loss_tokens": 17114827, "grad_norm": 2.984375, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 743.4, "dropped_truncated": 0, "gold_loss": 1.6655, "gold_lambda": 0.5, "rep_ratio": 2.439, "t_data_s": 0.0, "t_rollout_s": 57.6, "t_step_s": 132.8, "t_refresh_s": 0.3, "mem_gb": 10.55, "mem_gb_teacher": 20.67}
|
| 103 |
+
{"step": 98, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.589147219391016, "tokens": 152269, "cumulative_loss_tokens": 17267096, "grad_norm": 3.453125, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 594.8, "dropped_truncated": 0, "gold_loss": 1.5542, "gold_lambda": 0.5, "rep_ratio": 2.354, "t_data_s": 0.0, "t_rollout_s": 48.0, "t_step_s": 119.8, "t_refresh_s": 0.3, "mem_gb": 10.3, "mem_gb_teacher": 20.57}
|
| 104 |
+
{"step": 99, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.48909174772043035, "tokens": 175760, "cumulative_loss_tokens": 17442856, "grad_norm": 1.875, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 686.6, "dropped_truncated": 0, "gold_loss": 0.4908, "gold_lambda": 0.5, "rep_ratio": 2.447, "t_data_s": 0.0, "t_rollout_s": 52.2, "t_step_s": 126.9, "t_refresh_s": 0.3, "mem_gb": 10.25, "mem_gb_teacher": 20.56}
|
| 105 |
+
{"step": 100, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5824684727047363, "tokens": 170574, "cumulative_loss_tokens": 17613430, "grad_norm": 2.65625, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 666.3, "dropped_truncated": 0, "gold_loss": 0.4887, "gold_lambda": 0.5, "rep_ratio": 2.527, "t_data_s": 0.0, "t_rollout_s": 50.6, "t_step_s": 123.5, "t_refresh_s": 0.3, "mem_gb": 10.37, "mem_gb_teacher": 20.53}
|
| 106 |
+
{"step": 100, "gsm8k_n": 256, "gsm8k_quick_chat": 0.5859375, "t_eval_s": 11.5}
|
| 107 |
+
{"step": 101, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5455907765832685, "tokens": 189985, "cumulative_loss_tokens": 17803415, "grad_norm": 2.921875, "lr": 3e-06, "finish_rate": 0.973, "comp_len": 742.1, "dropped_truncated": 0, "gold_loss": 1.6509, "gold_lambda": 0.5, "rep_ratio": 2.585, "t_data_s": 0.0, "t_rollout_s": 56.9, "t_step_s": 134.5, "t_refresh_s": 0.3, "mem_gb": 10.34, "mem_gb_teacher": 20.58}
|
| 108 |
+
{"step": 102, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6113095593118041, "tokens": 162766, "cumulative_loss_tokens": 17966181, "grad_norm": 3.046875, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 635.8, "dropped_truncated": 0, "gold_loss": 1.777, "gold_lambda": 0.5, "rep_ratio": 2.239, "t_data_s": 0.0, "t_rollout_s": 49.7, "t_step_s": 120.8, "t_refresh_s": 0.3, "mem_gb": 10.4, "mem_gb_teacher": 20.55}
|
| 109 |
+
{"step": 103, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5893852222266681, "tokens": 182286, "cumulative_loss_tokens": 18148467, "grad_norm": 3.09375, "lr": 3e-06, "finish_rate": 0.996, "comp_len": 712.1, "dropped_truncated": 0, "gold_loss": 0.4122, "gold_lambda": 0.5, "rep_ratio": 2.435, "t_data_s": 0.0, "t_rollout_s": 54.3, "t_step_s": 129.6, "t_refresh_s": 0.3, "mem_gb": 10.63, "mem_gb_teacher": 20.63}
|
| 110 |
+
{"step": 104, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4894855239402883, "tokens": 165667, "cumulative_loss_tokens": 18314134, "grad_norm": 2.203125, "lr": 3e-06, "finish_rate": 0.996, "comp_len": 647.1, "dropped_truncated": 0, "gold_loss": 0.4492, "gold_lambda": 0.5, "rep_ratio": 2.445, "t_data_s": 0.0, "t_rollout_s": 50.4, "t_step_s": 122.4, "t_refresh_s": 0.3, "mem_gb": 10.16, "mem_gb_teacher": 20.53}
|
| 111 |
+
{"step": 105, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6073816959606229, "tokens": 150425, "cumulative_loss_tokens": 18464559, "grad_norm": 3.703125, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 587.6, "dropped_truncated": 0, "gold_loss": 1.1934, "gold_lambda": 0.5, "rep_ratio": 2.314, "t_data_s": 0.0, "t_rollout_s": 46.7, "t_step_s": 115.0, "t_refresh_s": 0.3, "mem_gb": 10.12, "mem_gb_teacher": 20.52}
|
| 112 |
+
{"step": 106, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.655743916945871, "tokens": 167127, "cumulative_loss_tokens": 18631686, "grad_norm": 3.8125, "lr": 3e-06, "finish_rate": 0.996, "comp_len": 652.8, "dropped_truncated": 0, "gold_loss": 1.6213, "gold_lambda": 0.5, "rep_ratio": 2.632, "t_data_s": 0.0, "t_rollout_s": 52.7, "t_step_s": 127.1, "t_refresh_s": 0.3, "mem_gb": 10.52, "mem_gb_teacher": 20.59}
|
| 113 |
+
{"step": 107, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5306262824191685, "tokens": 163182, "cumulative_loss_tokens": 18794868, "grad_norm": 2.53125, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 637.4, "dropped_truncated": 0, "gold_loss": 1.6076, "gold_lambda": 0.5, "rep_ratio": 2.462, "t_data_s": 0.0, "t_rollout_s": 50.1, "t_step_s": 122.1, "t_refresh_s": 0.3, "mem_gb": 10.08, "mem_gb_teacher": 20.51}
|
| 114 |
+
{"step": 108, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5666536525873619, "tokens": 202170, "cumulative_loss_tokens": 18997038, "grad_norm": 3.015625, "lr": 3e-06, "finish_rate": 0.977, "comp_len": 789.7, "dropped_truncated": 0, "gold_loss": 1.7017, "gold_lambda": 0.5, "rep_ratio": 2.319, "t_data_s": 0.0, "t_rollout_s": 61.1, "t_step_s": 140.7, "t_refresh_s": 0.3, "mem_gb": 10.67, "mem_gb_teacher": 20.63}
|
| 115 |
+
{"step": 109, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5767456456468123, "tokens": 164832, "cumulative_loss_tokens": 19161870, "grad_norm": 3.40625, "lr": 3e-06, "finish_rate": 0.996, "comp_len": 643.9, "dropped_truncated": 0, "gold_loss": 1.2246, "gold_lambda": 0.5, "rep_ratio": 2.57, "t_data_s": 0.0, "t_rollout_s": 50.4, "t_step_s": 120.2, "t_refresh_s": 0.3, "mem_gb": 10.87, "mem_gb_teacher": 20.72}
|
| 116 |
+
{"step": 110, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5696655467128002, "tokens": 171469, "cumulative_loss_tokens": 19333339, "grad_norm": 3.71875, "lr": 3e-06, "finish_rate": 0.969, "comp_len": 669.8, "dropped_truncated": 0, "gold_loss": 1.4079, "gold_lambda": 0.5, "rep_ratio": 2.528, "t_data_s": 0.0, "t_rollout_s": 52.2, "t_step_s": 123.5, "t_refresh_s": 0.3, "mem_gb": 10.34, "mem_gb_teacher": 20.59}
|
| 117 |
+
{"step": 111, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5980168485714425, "tokens": 193326, "cumulative_loss_tokens": 19526665, "grad_norm": 3.671875, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 755.2, "dropped_truncated": 0, "gold_loss": 1.4988, "gold_lambda": 0.5, "rep_ratio": 2.32, "t_data_s": 0.0, "t_rollout_s": 57.1, "t_step_s": 132.7, "t_refresh_s": 0.3, "mem_gb": 10.32, "mem_gb_teacher": 20.59}
|
| 118 |
+
{"step": 112, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5018475987546118, "tokens": 188531, "cumulative_loss_tokens": 19715196, "grad_norm": 2.3125, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 736.4, "dropped_truncated": 0, "gold_loss": 1.7268, "gold_lambda": 0.5, "rep_ratio": 2.253, "t_data_s": 0.0, "t_rollout_s": 56.4, "t_step_s": 132.4, "t_refresh_s": 0.3, "mem_gb": 10.51, "mem_gb_teacher": 20.66}
|
| 119 |
+
{"step": 113, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5937434506750944, "tokens": 160918, "cumulative_loss_tokens": 19876114, "grad_norm": 3.140625, "lr": 3e-06, "finish_rate": 1.0, "comp_len": 628.6, "dropped_truncated": 0, "gold_loss": 1.2401, "gold_lambda": 0.5, "rep_ratio": 2.624, "t_data_s": 0.0, "t_rollout_s": 48.7, "t_step_s": 119.0, "t_refresh_s": 0.3, "mem_gb": 10.22, "mem_gb_teacher": 20.54}
|
| 120 |
+
{"step": 114, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6009174510144456, "tokens": 160606, "cumulative_loss_tokens": 20036720, "grad_norm": 3.59375, "lr": 3e-06, "finish_rate": 0.992, "comp_len": 627.4, "dropped_truncated": 0, "gold_loss": 1.2568, "gold_lambda": 0.5, "rep_ratio": 2.413, "t_data_s": 0.0, "t_rollout_s": 49.1, "t_step_s": 119.9, "t_refresh_s": 0.3, "mem_gb": 10.44, "mem_gb_teacher": 20.57}
|
| 121 |
+
{"step": 115, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6648006230381973, "tokens": 153993, "cumulative_loss_tokens": 20190713, "grad_norm": 3.28125, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 601.5, "dropped_truncated": 0, "gold_loss": 0.3747, "gold_lambda": 0.5, "rep_ratio": 2.433, "t_data_s": 0.0, "t_rollout_s": 48.1, "t_step_s": 117.1, "t_refresh_s": 0.3, "mem_gb": 10.29, "mem_gb_teacher": 20.54}
|
| 122 |
+
{"step": 116, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6193397262170726, "tokens": 174995, "cumulative_loss_tokens": 20365708, "grad_norm": 2.90625, "lr": 3e-06, "finish_rate": 0.98, "comp_len": 683.6, "dropped_truncated": 0, "gold_loss": 0.3709, "gold_lambda": 0.5, "rep_ratio": 2.591, "t_data_s": 0.0, "t_rollout_s": 52.2, "t_step_s": 127.6, "t_refresh_s": 0.3, "mem_gb": 10.38, "mem_gb_teacher": 20.61}
|
| 123 |
+
{"step": 117, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.599443940443399, "tokens": 166579, "cumulative_loss_tokens": 20532287, "grad_norm": 3.625, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 650.7, "dropped_truncated": 0, "gold_loss": 1.4511, "gold_lambda": 0.5, "rep_ratio": 2.647, "t_data_s": 0.0, "t_rollout_s": 51.8, "t_step_s": 124.6, "t_refresh_s": 0.3, "mem_gb": 10.17, "mem_gb_teacher": 20.53}
|
| 124 |
+
{"step": 118, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5830937056629717, "tokens": 152384, "cumulative_loss_tokens": 20684671, "grad_norm": 3.125, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 595.2, "dropped_truncated": 0, "gold_loss": 1.8184, "gold_lambda": 0.5, "rep_ratio": 2.395, "t_data_s": 0.0, "t_rollout_s": 47.4, "t_step_s": 117.2, "t_refresh_s": 0.3, "mem_gb": 10.62, "mem_gb_teacher": 20.63}
|
| 125 |
+
{"step": 119, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.45885672965887614, "tokens": 188284, "cumulative_loss_tokens": 20872955, "grad_norm": 2.25, "lr": 3e-06, "finish_rate": 0.984, "comp_len": 735.5, "dropped_truncated": 0, "gold_loss": 1.4819, "gold_lambda": 0.5, "rep_ratio": 2.686, "t_data_s": 0.0, "t_rollout_s": 55.8, "t_step_s": 132.1, "t_refresh_s": 0.3, "mem_gb": 10.33, "mem_gb_teacher": 20.59}
|
| 126 |
+
{"step": 120, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5992642945291111, "tokens": 164143, "cumulative_loss_tokens": 21037098, "grad_norm": 2.9375, "lr": 3e-06, "finish_rate": 0.988, "comp_len": 641.2, "dropped_truncated": 0, "gold_loss": 1.4487, "gold_lambda": 0.5, "rep_ratio": 2.582, "t_data_s": 0.0, "t_rollout_s": 49.7, "t_step_s": 110.1, "t_refresh_s": 0.0, "mem_gb": 10.5, "mem_gb_teacher": 20.59}
|
| 127 |
+
{"step": 120, "gsm8k_n": 256, "gsm8k_quick_chat": 0.58984375, "t_eval_s": 23.3}
|
healed/stableopd_cold_keep50/vllm_server.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/stableopd_cold_keep50/wandb_sync.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Find logs at: /tmp/debug-cli.henry.log
|
| 2 |
+
Syncing: https://wandb.ai/hbfreed/glean-heal/runs/jrbo8gac ... done.
|
pruned/glean-0125inst-math-keep50/chat_template.jinja
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{{ bos_token }}{% for message in messages %}{% if message['role'] == 'system' %}{{ '<|system|>
|
| 2 |
+
' + message['content'] + '
|
| 3 |
+
' }}{% elif message['role'] == 'user' %}{{ '<|user|>
|
| 4 |
+
' + message['content'] + '
|
| 5 |
+
' }}{% elif message['role'] == 'assistant' %}{% if not loop.last %}{{ '<|assistant|>
|
| 6 |
+
' + message['content'] + eos_token + '
|
| 7 |
+
' }}{% else %}{{ '<|assistant|>
|
| 8 |
+
' + message['content'] + eos_token }}{% endif %}{% endif %}{% if loop.last and add_generation_prompt %}{{ '<|assistant|>
|
| 9 |
+
' }}{% endif %}{% endfor %}
|
pruned/glean-0125inst-math-keep50/config.json
ADDED
|
@@ -0,0 +1,887 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"OlmoeForCausalLM"
|
| 4 |
+
],
|
| 5 |
+
"attention_bias": false,
|
| 6 |
+
"attention_dropout": 0.0,
|
| 7 |
+
"auto_map": {
|
| 8 |
+
"AutoConfig": "configuration_pruned_olmoe.PrunedOlmoeConfig",
|
| 9 |
+
"AutoModelForCausalLM": "modeling_pruned_olmoe.PrunedOlmoeForCausalLM"
|
| 10 |
+
},
|
| 11 |
+
"clip_qkv": null,
|
| 12 |
+
"dtype": "bfloat16",
|
| 13 |
+
"eos_token_id": 50279,
|
| 14 |
+
"expert_widths": [
|
| 15 |
+
[
|
| 16 |
+
1024,
|
| 17 |
+
384,
|
| 18 |
+
256,
|
| 19 |
+
256,
|
| 20 |
+
768,
|
| 21 |
+
1024,
|
| 22 |
+
128,
|
| 23 |
+
768,
|
| 24 |
+
384,
|
| 25 |
+
128,
|
| 26 |
+
1024,
|
| 27 |
+
384,
|
| 28 |
+
384,
|
| 29 |
+
128,
|
| 30 |
+
640,
|
| 31 |
+
896,
|
| 32 |
+
896,
|
| 33 |
+
256,
|
| 34 |
+
768,
|
| 35 |
+
128,
|
| 36 |
+
768,
|
| 37 |
+
768,
|
| 38 |
+
768,
|
| 39 |
+
768,
|
| 40 |
+
640,
|
| 41 |
+
896,
|
| 42 |
+
256,
|
| 43 |
+
128,
|
| 44 |
+
384,
|
| 45 |
+
896,
|
| 46 |
+
256,
|
| 47 |
+
768,
|
| 48 |
+
512,
|
| 49 |
+
640,
|
| 50 |
+
768,
|
| 51 |
+
896,
|
| 52 |
+
1024,
|
| 53 |
+
384,
|
| 54 |
+
384,
|
| 55 |
+
512,
|
| 56 |
+
384,
|
| 57 |
+
256,
|
| 58 |
+
768,
|
| 59 |
+
640,
|
| 60 |
+
896,
|
| 61 |
+
896,
|
| 62 |
+
512,
|
| 63 |
+
768,
|
| 64 |
+
512,
|
| 65 |
+
384,
|
| 66 |
+
1024,
|
| 67 |
+
896,
|
| 68 |
+
896,
|
| 69 |
+
768,
|
| 70 |
+
128
|
| 71 |
+
],
|
| 72 |
+
[
|
| 73 |
+
768,
|
| 74 |
+
896,
|
| 75 |
+
896,
|
| 76 |
+
256,
|
| 77 |
+
896,
|
| 78 |
+
384,
|
| 79 |
+
768,
|
| 80 |
+
768,
|
| 81 |
+
1024,
|
| 82 |
+
384,
|
| 83 |
+
512,
|
| 84 |
+
1024,
|
| 85 |
+
768,
|
| 86 |
+
768,
|
| 87 |
+
1024,
|
| 88 |
+
640,
|
| 89 |
+
128,
|
| 90 |
+
768,
|
| 91 |
+
384,
|
| 92 |
+
768,
|
| 93 |
+
128,
|
| 94 |
+
256,
|
| 95 |
+
512,
|
| 96 |
+
384,
|
| 97 |
+
896,
|
| 98 |
+
256,
|
| 99 |
+
1024,
|
| 100 |
+
1024,
|
| 101 |
+
896,
|
| 102 |
+
1024,
|
| 103 |
+
896,
|
| 104 |
+
768,
|
| 105 |
+
256,
|
| 106 |
+
768,
|
| 107 |
+
128,
|
| 108 |
+
1024,
|
| 109 |
+
768,
|
| 110 |
+
768,
|
| 111 |
+
640,
|
| 112 |
+
384,
|
| 113 |
+
1024,
|
| 114 |
+
384,
|
| 115 |
+
640,
|
| 116 |
+
896,
|
| 117 |
+
1024,
|
| 118 |
+
512,
|
| 119 |
+
512,
|
| 120 |
+
512,
|
| 121 |
+
640
|
| 122 |
+
],
|
| 123 |
+
[
|
| 124 |
+
768,
|
| 125 |
+
1024,
|
| 126 |
+
512,
|
| 127 |
+
384,
|
| 128 |
+
384,
|
| 129 |
+
768,
|
| 130 |
+
256,
|
| 131 |
+
384,
|
| 132 |
+
1024,
|
| 133 |
+
896,
|
| 134 |
+
640,
|
| 135 |
+
256,
|
| 136 |
+
1024,
|
| 137 |
+
1024,
|
| 138 |
+
1024,
|
| 139 |
+
640,
|
| 140 |
+
1024,
|
| 141 |
+
896,
|
| 142 |
+
512,
|
| 143 |
+
768,
|
| 144 |
+
128,
|
| 145 |
+
768,
|
| 146 |
+
1024,
|
| 147 |
+
768,
|
| 148 |
+
256,
|
| 149 |
+
1024,
|
| 150 |
+
896,
|
| 151 |
+
1024,
|
| 152 |
+
896,
|
| 153 |
+
512,
|
| 154 |
+
640,
|
| 155 |
+
384,
|
| 156 |
+
896,
|
| 157 |
+
384,
|
| 158 |
+
512,
|
| 159 |
+
640,
|
| 160 |
+
256,
|
| 161 |
+
768,
|
| 162 |
+
256,
|
| 163 |
+
768,
|
| 164 |
+
768,
|
| 165 |
+
640,
|
| 166 |
+
1024,
|
| 167 |
+
640,
|
| 168 |
+
896,
|
| 169 |
+
768,
|
| 170 |
+
768,
|
| 171 |
+
256
|
| 172 |
+
],
|
| 173 |
+
[
|
| 174 |
+
640,
|
| 175 |
+
640,
|
| 176 |
+
896,
|
| 177 |
+
512,
|
| 178 |
+
640,
|
| 179 |
+
896,
|
| 180 |
+
896,
|
| 181 |
+
384,
|
| 182 |
+
896,
|
| 183 |
+
896,
|
| 184 |
+
384,
|
| 185 |
+
896,
|
| 186 |
+
640,
|
| 187 |
+
256,
|
| 188 |
+
640,
|
| 189 |
+
1024,
|
| 190 |
+
1024,
|
| 191 |
+
768,
|
| 192 |
+
1024,
|
| 193 |
+
896,
|
| 194 |
+
1024,
|
| 195 |
+
768,
|
| 196 |
+
896,
|
| 197 |
+
256,
|
| 198 |
+
512,
|
| 199 |
+
768,
|
| 200 |
+
1024,
|
| 201 |
+
256,
|
| 202 |
+
768,
|
| 203 |
+
512,
|
| 204 |
+
256,
|
| 205 |
+
640,
|
| 206 |
+
1024,
|
| 207 |
+
1024,
|
| 208 |
+
512,
|
| 209 |
+
1024,
|
| 210 |
+
768,
|
| 211 |
+
256,
|
| 212 |
+
1024,
|
| 213 |
+
384,
|
| 214 |
+
896,
|
| 215 |
+
1024,
|
| 216 |
+
896,
|
| 217 |
+
1024,
|
| 218 |
+
384
|
| 219 |
+
],
|
| 220 |
+
[
|
| 221 |
+
256,
|
| 222 |
+
640,
|
| 223 |
+
640,
|
| 224 |
+
896,
|
| 225 |
+
768,
|
| 226 |
+
896,
|
| 227 |
+
768,
|
| 228 |
+
768,
|
| 229 |
+
896,
|
| 230 |
+
896,
|
| 231 |
+
1024,
|
| 232 |
+
256,
|
| 233 |
+
512,
|
| 234 |
+
1024,
|
| 235 |
+
640,
|
| 236 |
+
896,
|
| 237 |
+
512,
|
| 238 |
+
512,
|
| 239 |
+
384,
|
| 240 |
+
384,
|
| 241 |
+
256,
|
| 242 |
+
384,
|
| 243 |
+
384,
|
| 244 |
+
896,
|
| 245 |
+
896,
|
| 246 |
+
768,
|
| 247 |
+
640,
|
| 248 |
+
896,
|
| 249 |
+
768,
|
| 250 |
+
1024,
|
| 251 |
+
512,
|
| 252 |
+
640,
|
| 253 |
+
512,
|
| 254 |
+
640,
|
| 255 |
+
896,
|
| 256 |
+
512,
|
| 257 |
+
512,
|
| 258 |
+
384,
|
| 259 |
+
640,
|
| 260 |
+
896,
|
| 261 |
+
896,
|
| 262 |
+
896,
|
| 263 |
+
1024,
|
| 264 |
+
640,
|
| 265 |
+
1024,
|
| 266 |
+
640,
|
| 267 |
+
1024
|
| 268 |
+
],
|
| 269 |
+
[
|
| 270 |
+
1024,
|
| 271 |
+
512,
|
| 272 |
+
1024,
|
| 273 |
+
1024,
|
| 274 |
+
640,
|
| 275 |
+
896,
|
| 276 |
+
640,
|
| 277 |
+
1024,
|
| 278 |
+
896,
|
| 279 |
+
384,
|
| 280 |
+
1024,
|
| 281 |
+
128,
|
| 282 |
+
896,
|
| 283 |
+
768,
|
| 284 |
+
1024,
|
| 285 |
+
768,
|
| 286 |
+
640,
|
| 287 |
+
896,
|
| 288 |
+
768,
|
| 289 |
+
640,
|
| 290 |
+
512,
|
| 291 |
+
896,
|
| 292 |
+
512,
|
| 293 |
+
640,
|
| 294 |
+
256,
|
| 295 |
+
768,
|
| 296 |
+
640,
|
| 297 |
+
768,
|
| 298 |
+
384,
|
| 299 |
+
896,
|
| 300 |
+
512,
|
| 301 |
+
512,
|
| 302 |
+
256,
|
| 303 |
+
512,
|
| 304 |
+
896,
|
| 305 |
+
256,
|
| 306 |
+
384,
|
| 307 |
+
640,
|
| 308 |
+
512,
|
| 309 |
+
640,
|
| 310 |
+
896,
|
| 311 |
+
512,
|
| 312 |
+
1024,
|
| 313 |
+
256,
|
| 314 |
+
768,
|
| 315 |
+
1024,
|
| 316 |
+
768,
|
| 317 |
+
256,
|
| 318 |
+
256
|
| 319 |
+
],
|
| 320 |
+
[
|
| 321 |
+
640,
|
| 322 |
+
896,
|
| 323 |
+
1024,
|
| 324 |
+
896,
|
| 325 |
+
1024,
|
| 326 |
+
1024,
|
| 327 |
+
1024,
|
| 328 |
+
512,
|
| 329 |
+
256,
|
| 330 |
+
256,
|
| 331 |
+
1024,
|
| 332 |
+
768,
|
| 333 |
+
512,
|
| 334 |
+
768,
|
| 335 |
+
1024,
|
| 336 |
+
1024,
|
| 337 |
+
1024,
|
| 338 |
+
384,
|
| 339 |
+
512,
|
| 340 |
+
1024,
|
| 341 |
+
512,
|
| 342 |
+
1024,
|
| 343 |
+
128,
|
| 344 |
+
640,
|
| 345 |
+
640,
|
| 346 |
+
896,
|
| 347 |
+
768,
|
| 348 |
+
128,
|
| 349 |
+
256,
|
| 350 |
+
256,
|
| 351 |
+
256,
|
| 352 |
+
256,
|
| 353 |
+
896,
|
| 354 |
+
1024,
|
| 355 |
+
1024,
|
| 356 |
+
384,
|
| 357 |
+
896,
|
| 358 |
+
256,
|
| 359 |
+
896,
|
| 360 |
+
640,
|
| 361 |
+
1024,
|
| 362 |
+
384,
|
| 363 |
+
640,
|
| 364 |
+
256,
|
| 365 |
+
1024,
|
| 366 |
+
1024,
|
| 367 |
+
1024
|
| 368 |
+
],
|
| 369 |
+
[
|
| 370 |
+
1024,
|
| 371 |
+
384,
|
| 372 |
+
1024,
|
| 373 |
+
1024,
|
| 374 |
+
256,
|
| 375 |
+
128,
|
| 376 |
+
256,
|
| 377 |
+
384,
|
| 378 |
+
256,
|
| 379 |
+
384,
|
| 380 |
+
896,
|
| 381 |
+
768,
|
| 382 |
+
896,
|
| 383 |
+
896,
|
| 384 |
+
512,
|
| 385 |
+
896,
|
| 386 |
+
640,
|
| 387 |
+
384,
|
| 388 |
+
384,
|
| 389 |
+
896,
|
| 390 |
+
768,
|
| 391 |
+
384,
|
| 392 |
+
896,
|
| 393 |
+
768,
|
| 394 |
+
768,
|
| 395 |
+
512,
|
| 396 |
+
896,
|
| 397 |
+
768,
|
| 398 |
+
768,
|
| 399 |
+
896,
|
| 400 |
+
128,
|
| 401 |
+
896,
|
| 402 |
+
512,
|
| 403 |
+
256,
|
| 404 |
+
768,
|
| 405 |
+
128,
|
| 406 |
+
384,
|
| 407 |
+
256,
|
| 408 |
+
896,
|
| 409 |
+
896,
|
| 410 |
+
384,
|
| 411 |
+
768,
|
| 412 |
+
512,
|
| 413 |
+
640,
|
| 414 |
+
256,
|
| 415 |
+
768,
|
| 416 |
+
640,
|
| 417 |
+
896,
|
| 418 |
+
384,
|
| 419 |
+
512,
|
| 420 |
+
1024,
|
| 421 |
+
768,
|
| 422 |
+
384
|
| 423 |
+
],
|
| 424 |
+
[
|
| 425 |
+
512,
|
| 426 |
+
768,
|
| 427 |
+
512,
|
| 428 |
+
256,
|
| 429 |
+
128,
|
| 430 |
+
640,
|
| 431 |
+
384,
|
| 432 |
+
640,
|
| 433 |
+
768,
|
| 434 |
+
896,
|
| 435 |
+
640,
|
| 436 |
+
768,
|
| 437 |
+
256,
|
| 438 |
+
384,
|
| 439 |
+
1024,
|
| 440 |
+
896,
|
| 441 |
+
256,
|
| 442 |
+
896,
|
| 443 |
+
512,
|
| 444 |
+
256,
|
| 445 |
+
896,
|
| 446 |
+
768,
|
| 447 |
+
256,
|
| 448 |
+
896,
|
| 449 |
+
896,
|
| 450 |
+
384,
|
| 451 |
+
896,
|
| 452 |
+
640,
|
| 453 |
+
768,
|
| 454 |
+
512,
|
| 455 |
+
768,
|
| 456 |
+
768,
|
| 457 |
+
1024,
|
| 458 |
+
768,
|
| 459 |
+
640,
|
| 460 |
+
768,
|
| 461 |
+
384,
|
| 462 |
+
256,
|
| 463 |
+
512,
|
| 464 |
+
896,
|
| 465 |
+
128,
|
| 466 |
+
384,
|
| 467 |
+
256,
|
| 468 |
+
768,
|
| 469 |
+
384,
|
| 470 |
+
256,
|
| 471 |
+
1024,
|
| 472 |
+
1024,
|
| 473 |
+
896,
|
| 474 |
+
256,
|
| 475 |
+
1024,
|
| 476 |
+
256,
|
| 477 |
+
128,
|
| 478 |
+
896
|
| 479 |
+
],
|
| 480 |
+
[
|
| 481 |
+
640,
|
| 482 |
+
640,
|
| 483 |
+
896,
|
| 484 |
+
256,
|
| 485 |
+
1024,
|
| 486 |
+
512,
|
| 487 |
+
1024,
|
| 488 |
+
768,
|
| 489 |
+
384,
|
| 490 |
+
512,
|
| 491 |
+
256,
|
| 492 |
+
768,
|
| 493 |
+
896,
|
| 494 |
+
768,
|
| 495 |
+
512,
|
| 496 |
+
768,
|
| 497 |
+
768,
|
| 498 |
+
640,
|
| 499 |
+
384,
|
| 500 |
+
768,
|
| 501 |
+
512,
|
| 502 |
+
768,
|
| 503 |
+
768,
|
| 504 |
+
512,
|
| 505 |
+
768,
|
| 506 |
+
128,
|
| 507 |
+
896,
|
| 508 |
+
512,
|
| 509 |
+
768,
|
| 510 |
+
1024,
|
| 511 |
+
128,
|
| 512 |
+
384,
|
| 513 |
+
768,
|
| 514 |
+
768,
|
| 515 |
+
768,
|
| 516 |
+
384,
|
| 517 |
+
512,
|
| 518 |
+
640,
|
| 519 |
+
768,
|
| 520 |
+
512,
|
| 521 |
+
768,
|
| 522 |
+
1024,
|
| 523 |
+
640,
|
| 524 |
+
896,
|
| 525 |
+
256,
|
| 526 |
+
1024,
|
| 527 |
+
384,
|
| 528 |
+
768,
|
| 529 |
+
768,
|
| 530 |
+
768
|
| 531 |
+
],
|
| 532 |
+
[
|
| 533 |
+
896,
|
| 534 |
+
512,
|
| 535 |
+
896,
|
| 536 |
+
768,
|
| 537 |
+
384,
|
| 538 |
+
384,
|
| 539 |
+
768,
|
| 540 |
+
512,
|
| 541 |
+
768,
|
| 542 |
+
512,
|
| 543 |
+
1024,
|
| 544 |
+
640,
|
| 545 |
+
896,
|
| 546 |
+
896,
|
| 547 |
+
256,
|
| 548 |
+
640,
|
| 549 |
+
1024,
|
| 550 |
+
256,
|
| 551 |
+
896,
|
| 552 |
+
128,
|
| 553 |
+
128,
|
| 554 |
+
128,
|
| 555 |
+
768,
|
| 556 |
+
896,
|
| 557 |
+
384,
|
| 558 |
+
896,
|
| 559 |
+
512,
|
| 560 |
+
896,
|
| 561 |
+
384,
|
| 562 |
+
256,
|
| 563 |
+
640,
|
| 564 |
+
640,
|
| 565 |
+
896,
|
| 566 |
+
768,
|
| 567 |
+
640,
|
| 568 |
+
256,
|
| 569 |
+
896,
|
| 570 |
+
896,
|
| 571 |
+
512,
|
| 572 |
+
128,
|
| 573 |
+
896,
|
| 574 |
+
256,
|
| 575 |
+
256,
|
| 576 |
+
640,
|
| 577 |
+
896,
|
| 578 |
+
896,
|
| 579 |
+
128,
|
| 580 |
+
1024,
|
| 581 |
+
256,
|
| 582 |
+
384,
|
| 583 |
+
1024,
|
| 584 |
+
640,
|
| 585 |
+
896
|
| 586 |
+
],
|
| 587 |
+
[
|
| 588 |
+
768,
|
| 589 |
+
384,
|
| 590 |
+
640,
|
| 591 |
+
896,
|
| 592 |
+
256,
|
| 593 |
+
128,
|
| 594 |
+
384,
|
| 595 |
+
896,
|
| 596 |
+
128,
|
| 597 |
+
128,
|
| 598 |
+
896,
|
| 599 |
+
256,
|
| 600 |
+
384,
|
| 601 |
+
896,
|
| 602 |
+
512,
|
| 603 |
+
768,
|
| 604 |
+
768,
|
| 605 |
+
512,
|
| 606 |
+
512,
|
| 607 |
+
768,
|
| 608 |
+
896,
|
| 609 |
+
640,
|
| 610 |
+
768,
|
| 611 |
+
896,
|
| 612 |
+
896,
|
| 613 |
+
896,
|
| 614 |
+
640,
|
| 615 |
+
896,
|
| 616 |
+
640,
|
| 617 |
+
512,
|
| 618 |
+
896,
|
| 619 |
+
256,
|
| 620 |
+
512,
|
| 621 |
+
128,
|
| 622 |
+
512,
|
| 623 |
+
384,
|
| 624 |
+
768,
|
| 625 |
+
768,
|
| 626 |
+
1024,
|
| 627 |
+
256,
|
| 628 |
+
768,
|
| 629 |
+
256,
|
| 630 |
+
768,
|
| 631 |
+
512,
|
| 632 |
+
640,
|
| 633 |
+
1024,
|
| 634 |
+
128,
|
| 635 |
+
896,
|
| 636 |
+
896,
|
| 637 |
+
896,
|
| 638 |
+
896,
|
| 639 |
+
1024
|
| 640 |
+
],
|
| 641 |
+
[
|
| 642 |
+
512,
|
| 643 |
+
384,
|
| 644 |
+
768,
|
| 645 |
+
640,
|
| 646 |
+
640,
|
| 647 |
+
768,
|
| 648 |
+
1024,
|
| 649 |
+
896,
|
| 650 |
+
512,
|
| 651 |
+
256,
|
| 652 |
+
640,
|
| 653 |
+
768,
|
| 654 |
+
640,
|
| 655 |
+
896,
|
| 656 |
+
128,
|
| 657 |
+
256,
|
| 658 |
+
896,
|
| 659 |
+
1024,
|
| 660 |
+
256,
|
| 661 |
+
640,
|
| 662 |
+
512,
|
| 663 |
+
256,
|
| 664 |
+
128,
|
| 665 |
+
512,
|
| 666 |
+
256,
|
| 667 |
+
640,
|
| 668 |
+
768,
|
| 669 |
+
768,
|
| 670 |
+
128,
|
| 671 |
+
128,
|
| 672 |
+
768,
|
| 673 |
+
640,
|
| 674 |
+
1024,
|
| 675 |
+
1024,
|
| 676 |
+
768,
|
| 677 |
+
512,
|
| 678 |
+
896,
|
| 679 |
+
768,
|
| 680 |
+
896,
|
| 681 |
+
1024,
|
| 682 |
+
896,
|
| 683 |
+
896,
|
| 684 |
+
512,
|
| 685 |
+
640,
|
| 686 |
+
1024,
|
| 687 |
+
512,
|
| 688 |
+
1024,
|
| 689 |
+
512,
|
| 690 |
+
512,
|
| 691 |
+
512,
|
| 692 |
+
768
|
| 693 |
+
],
|
| 694 |
+
[
|
| 695 |
+
896,
|
| 696 |
+
896,
|
| 697 |
+
1024,
|
| 698 |
+
1024,
|
| 699 |
+
896,
|
| 700 |
+
128,
|
| 701 |
+
768,
|
| 702 |
+
256,
|
| 703 |
+
1024,
|
| 704 |
+
256,
|
| 705 |
+
1024,
|
| 706 |
+
640,
|
| 707 |
+
384,
|
| 708 |
+
256,
|
| 709 |
+
256,
|
| 710 |
+
512,
|
| 711 |
+
768,
|
| 712 |
+
896,
|
| 713 |
+
512,
|
| 714 |
+
768,
|
| 715 |
+
384,
|
| 716 |
+
1024,
|
| 717 |
+
896,
|
| 718 |
+
896,
|
| 719 |
+
1024,
|
| 720 |
+
896,
|
| 721 |
+
768,
|
| 722 |
+
896,
|
| 723 |
+
640,
|
| 724 |
+
1024,
|
| 725 |
+
512,
|
| 726 |
+
896,
|
| 727 |
+
512,
|
| 728 |
+
1024,
|
| 729 |
+
512,
|
| 730 |
+
512,
|
| 731 |
+
256,
|
| 732 |
+
256,
|
| 733 |
+
256,
|
| 734 |
+
512,
|
| 735 |
+
768,
|
| 736 |
+
128,
|
| 737 |
+
384,
|
| 738 |
+
512,
|
| 739 |
+
896,
|
| 740 |
+
896,
|
| 741 |
+
1024,
|
| 742 |
+
256,
|
| 743 |
+
384,
|
| 744 |
+
640
|
| 745 |
+
],
|
| 746 |
+
[
|
| 747 |
+
896,
|
| 748 |
+
640,
|
| 749 |
+
384,
|
| 750 |
+
512,
|
| 751 |
+
256,
|
| 752 |
+
640,
|
| 753 |
+
1024,
|
| 754 |
+
384,
|
| 755 |
+
1024,
|
| 756 |
+
1024,
|
| 757 |
+
768,
|
| 758 |
+
256,
|
| 759 |
+
1024,
|
| 760 |
+
768,
|
| 761 |
+
512,
|
| 762 |
+
896,
|
| 763 |
+
256,
|
| 764 |
+
1024,
|
| 765 |
+
768,
|
| 766 |
+
768,
|
| 767 |
+
768,
|
| 768 |
+
384,
|
| 769 |
+
384,
|
| 770 |
+
256,
|
| 771 |
+
1024,
|
| 772 |
+
384,
|
| 773 |
+
384,
|
| 774 |
+
384,
|
| 775 |
+
896,
|
| 776 |
+
768,
|
| 777 |
+
640,
|
| 778 |
+
768,
|
| 779 |
+
512,
|
| 780 |
+
896,
|
| 781 |
+
896,
|
| 782 |
+
896,
|
| 783 |
+
896,
|
| 784 |
+
256,
|
| 785 |
+
384,
|
| 786 |
+
128,
|
| 787 |
+
1024,
|
| 788 |
+
896,
|
| 789 |
+
256,
|
| 790 |
+
256,
|
| 791 |
+
768,
|
| 792 |
+
640,
|
| 793 |
+
896,
|
| 794 |
+
384,
|
| 795 |
+
768,
|
| 796 |
+
512,
|
| 797 |
+
640
|
| 798 |
+
],
|
| 799 |
+
[
|
| 800 |
+
896,
|
| 801 |
+
1024,
|
| 802 |
+
768,
|
| 803 |
+
1024,
|
| 804 |
+
896,
|
| 805 |
+
256,
|
| 806 |
+
768,
|
| 807 |
+
128,
|
| 808 |
+
128,
|
| 809 |
+
768,
|
| 810 |
+
512,
|
| 811 |
+
896,
|
| 812 |
+
384,
|
| 813 |
+
768,
|
| 814 |
+
1024,
|
| 815 |
+
256,
|
| 816 |
+
768,
|
| 817 |
+
768,
|
| 818 |
+
256,
|
| 819 |
+
512,
|
| 820 |
+
512,
|
| 821 |
+
640,
|
| 822 |
+
512,
|
| 823 |
+
256,
|
| 824 |
+
768,
|
| 825 |
+
896,
|
| 826 |
+
384,
|
| 827 |
+
1024,
|
| 828 |
+
640,
|
| 829 |
+
1024,
|
| 830 |
+
512,
|
| 831 |
+
512,
|
| 832 |
+
384,
|
| 833 |
+
512,
|
| 834 |
+
512,
|
| 835 |
+
1024,
|
| 836 |
+
384,
|
| 837 |
+
896,
|
| 838 |
+
768,
|
| 839 |
+
384,
|
| 840 |
+
384,
|
| 841 |
+
128,
|
| 842 |
+
384,
|
| 843 |
+
1024,
|
| 844 |
+
896,
|
| 845 |
+
640,
|
| 846 |
+
768,
|
| 847 |
+
768,
|
| 848 |
+
256,
|
| 849 |
+
640,
|
| 850 |
+
512,
|
| 851 |
+
640,
|
| 852 |
+
384
|
| 853 |
+
]
|
| 854 |
+
],
|
| 855 |
+
"glean_metadata": {
|
| 856 |
+
"base_model": "allenai/OLMoE-1B-7B-0125-Instruct",
|
| 857 |
+
"block_size": 128,
|
| 858 |
+
"criterion": "reap",
|
| 859 |
+
"dead_experts": 217,
|
| 860 |
+
"keep_fraction": 0.5,
|
| 861 |
+
"min_width": 128,
|
| 862 |
+
"params": 3697491968,
|
| 863 |
+
"scores": "outputs/scores_0125inst_dolmino-math/scores.pt"
|
| 864 |
+
},
|
| 865 |
+
"hidden_act": "silu",
|
| 866 |
+
"hidden_size": 2048,
|
| 867 |
+
"initializer_range": 0.02,
|
| 868 |
+
"intermediate_size": 1024,
|
| 869 |
+
"max_position_embeddings": 4096,
|
| 870 |
+
"model_type": "pruned_olmoe",
|
| 871 |
+
"norm_topk_prob": false,
|
| 872 |
+
"num_attention_heads": 16,
|
| 873 |
+
"num_experts": 64,
|
| 874 |
+
"num_experts_per_tok": 8,
|
| 875 |
+
"num_hidden_layers": 16,
|
| 876 |
+
"num_key_value_heads": 16,
|
| 877 |
+
"output_router_logits": false,
|
| 878 |
+
"pad_token_id": 1,
|
| 879 |
+
"rms_norm_eps": 1e-05,
|
| 880 |
+
"rope_scaling": null,
|
| 881 |
+
"rope_theta": 10000.0,
|
| 882 |
+
"router_aux_loss_coef": 0.01,
|
| 883 |
+
"tie_word_embeddings": false,
|
| 884 |
+
"transformers_version": "4.57.6",
|
| 885 |
+
"use_cache": false,
|
| 886 |
+
"vocab_size": 50304
|
| 887 |
+
}
|
pruned/glean-0125inst-math-keep50/configuration_pruned_olmoe.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Configuration for GLEAN-pruned OLMoE: variable-width, variable-count experts.
|
| 2 |
+
|
| 3 |
+
Ships alongside modeling_pruned_olmoe.py inside every pruned checkpoint so it
|
| 4 |
+
loads with ``AutoModelForCausalLM.from_pretrained(..., trust_remote_code=True)``.
|
| 5 |
+
Standalone by design: imports transformers only.
|
| 6 |
+
"""
|
| 7 |
+
|
| 8 |
+
from transformers.models.olmoe.configuration_olmoe import OlmoeConfig
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
class PrunedOlmoeConfig(OlmoeConfig):
|
| 12 |
+
"""OlmoeConfig plus a per-(layer, expert) width table.
|
| 13 |
+
|
| 14 |
+
``expert_widths[l]`` lists the SwiGLU intermediate width of each surviving
|
| 15 |
+
expert in decoder layer ``l``, in expert order. Lists are ragged: layers
|
| 16 |
+
may keep different numbers of experts (deleted experts simply don't
|
| 17 |
+
appear — the router in layer ``l`` has ``len(expert_widths[l])`` rows),
|
| 18 |
+
and each width may differ (multiples of the GEMM block size, 128, for
|
| 19 |
+
variable-MegaBlocks execution). ``None`` means an unpruned model
|
| 20 |
+
(uniform ``num_experts`` × ``intermediate_size``).
|
| 21 |
+
|
| 22 |
+
The inherited ``num_experts`` / ``intermediate_size`` keep their ORIGINAL
|
| 23 |
+
(pre-pruning) values for provenance; the width table is authoritative for
|
| 24 |
+
the built architecture.
|
| 25 |
+
"""
|
| 26 |
+
|
| 27 |
+
model_type = "pruned_olmoe"
|
| 28 |
+
|
| 29 |
+
def __init__(self, expert_widths: list[list[int]] | None = None, **kwargs):
|
| 30 |
+
super().__init__(**kwargs)
|
| 31 |
+
self.expert_widths = expert_widths
|
pruned/glean-0125inst-math-keep50/generation_config.json
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_from_model_config": true,
|
| 3 |
+
"eos_token_id": 50279,
|
| 4 |
+
"pad_token_id": 1,
|
| 5 |
+
"transformers_version": "4.57.6"
|
| 6 |
+
}
|
pruned/glean-0125inst-math-keep50/model.safetensors.index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
pruned/glean-0125inst-math-keep50/modeling_pruned_olmoe.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""GLEAN-pruned OLMoE: HF-loadable model with ragged (variable-width) experts.
|
| 2 |
+
|
| 3 |
+
Pattern follows hbfreed/variable-flex-olmo's PrunedFlexOlmoForCausalLM
|
| 4 |
+
(docs/recon/prior-work-hbfreed.md), generalized from one scalar width to a
|
| 5 |
+
per-(layer, expert) width table: ``super().__init__`` builds the uniform
|
| 6 |
+
architecture from the config, then every MoE block is rebuilt to its pruned
|
| 7 |
+
shape — surviving experts only, each at its own width, router sliced to
|
| 8 |
+
match — so the state dict aligns exactly with what
|
| 9 |
+
``glean.prune.prune_channels_global`` leaves behind.
|
| 10 |
+
|
| 11 |
+
Caveat: ``output_router_logits=True`` (the load-balancing aux loss) assumes a
|
| 12 |
+
uniform ``config.num_experts`` and is unsupported on ragged models.
|
| 13 |
+
"""
|
| 14 |
+
|
| 15 |
+
import torch.nn as nn
|
| 16 |
+
from transformers.activations import ACT2FN
|
| 17 |
+
from transformers.models.olmoe.modeling_olmoe import OlmoeForCausalLM
|
| 18 |
+
|
| 19 |
+
from .configuration_pruned_olmoe import PrunedOlmoeConfig
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
class RaggedOlmoeMLP(nn.Module):
|
| 23 |
+
"""OlmoeMLP with an explicit intermediate width (SwiGLU, no biases)."""
|
| 24 |
+
|
| 25 |
+
def __init__(self, hidden_size: int, intermediate_size: int, hidden_act: str):
|
| 26 |
+
super().__init__()
|
| 27 |
+
self.hidden_size = hidden_size
|
| 28 |
+
self.intermediate_size = intermediate_size
|
| 29 |
+
self.gate_proj = nn.Linear(hidden_size, intermediate_size, bias=False)
|
| 30 |
+
self.up_proj = nn.Linear(hidden_size, intermediate_size, bias=False)
|
| 31 |
+
self.down_proj = nn.Linear(intermediate_size, hidden_size, bias=False)
|
| 32 |
+
self.act_fn = ACT2FN[hidden_act]
|
| 33 |
+
|
| 34 |
+
def forward(self, x):
|
| 35 |
+
return self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 36 |
+
|
| 37 |
+
|
| 38 |
+
class PrunedOlmoeForCausalLM(OlmoeForCausalLM):
|
| 39 |
+
"""OLMoE with per-layer surviving-expert lists at per-expert widths."""
|
| 40 |
+
|
| 41 |
+
config_class = PrunedOlmoeConfig
|
| 42 |
+
|
| 43 |
+
def __init__(self, config: PrunedOlmoeConfig):
|
| 44 |
+
super().__init__(config)
|
| 45 |
+
widths_table = getattr(config, "expert_widths", None)
|
| 46 |
+
if widths_table is None:
|
| 47 |
+
return # unpruned: plain OLMoE
|
| 48 |
+
if len(widths_table) != len(self.model.layers):
|
| 49 |
+
raise ValueError(
|
| 50 |
+
f"expert_widths has {len(widths_table)} rows but the model has "
|
| 51 |
+
f"{len(self.model.layers)} decoder layers"
|
| 52 |
+
)
|
| 53 |
+
for layer, widths in zip(self.model.layers, widths_table):
|
| 54 |
+
if any(w <= 0 for w in widths):
|
| 55 |
+
raise ValueError("expert_widths must list surviving experts only (>0)")
|
| 56 |
+
block = layer.mlp
|
| 57 |
+
if len(widths) < block.top_k:
|
| 58 |
+
raise ValueError(
|
| 59 |
+
f"a layer keeps {len(widths)} experts < top_k={block.top_k}"
|
| 60 |
+
)
|
| 61 |
+
block.num_experts = len(widths)
|
| 62 |
+
block.gate = nn.Linear(config.hidden_size, len(widths), bias=False)
|
| 63 |
+
block.experts = nn.ModuleList(
|
| 64 |
+
RaggedOlmoeMLP(config.hidden_size, w, config.hidden_act)
|
| 65 |
+
for w in widths
|
| 66 |
+
)
|
pruned/glean-0125inst-math-keep50/special_tokens_map.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": {
|
| 3 |
+
"content": "|||IP_ADDRESS|||",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": true,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"eos_token": {
|
| 10 |
+
"content": "|||IP_ADDRESS|||",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": true,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"pad_token": {
|
| 17 |
+
"content": "<pad>",
|
| 18 |
+
"lstrip": false,
|
| 19 |
+
"normalized": false,
|
| 20 |
+
"rstrip": false,
|
| 21 |
+
"single_word": false
|
| 22 |
+
}
|
| 23 |
+
}
|
pruned/glean-0125inst-math-keep50/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
pruned/glean-0125inst-math-keep50/tokenizer_config.json
ADDED
|
@@ -0,0 +1,247 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_bos_token": false,
|
| 3 |
+
"add_eos_token": false,
|
| 4 |
+
"add_prefix_space": false,
|
| 5 |
+
"added_tokens_decoder": {
|
| 6 |
+
"0": {
|
| 7 |
+
"content": "<|endoftext|>",
|
| 8 |
+
"lstrip": false,
|
| 9 |
+
"normalized": false,
|
| 10 |
+
"rstrip": false,
|
| 11 |
+
"single_word": false,
|
| 12 |
+
"special": true
|
| 13 |
+
},
|
| 14 |
+
"1": {
|
| 15 |
+
"content": "<|padding|>",
|
| 16 |
+
"lstrip": false,
|
| 17 |
+
"normalized": false,
|
| 18 |
+
"rstrip": false,
|
| 19 |
+
"single_word": false,
|
| 20 |
+
"special": true
|
| 21 |
+
},
|
| 22 |
+
"50254": {
|
| 23 |
+
"content": " ",
|
| 24 |
+
"lstrip": false,
|
| 25 |
+
"normalized": true,
|
| 26 |
+
"rstrip": false,
|
| 27 |
+
"single_word": false,
|
| 28 |
+
"special": false
|
| 29 |
+
},
|
| 30 |
+
"50255": {
|
| 31 |
+
"content": " ",
|
| 32 |
+
"lstrip": false,
|
| 33 |
+
"normalized": true,
|
| 34 |
+
"rstrip": false,
|
| 35 |
+
"single_word": false,
|
| 36 |
+
"special": false
|
| 37 |
+
},
|
| 38 |
+
"50256": {
|
| 39 |
+
"content": " ",
|
| 40 |
+
"lstrip": false,
|
| 41 |
+
"normalized": true,
|
| 42 |
+
"rstrip": false,
|
| 43 |
+
"single_word": false,
|
| 44 |
+
"special": false
|
| 45 |
+
},
|
| 46 |
+
"50257": {
|
| 47 |
+
"content": " ",
|
| 48 |
+
"lstrip": false,
|
| 49 |
+
"normalized": true,
|
| 50 |
+
"rstrip": false,
|
| 51 |
+
"single_word": false,
|
| 52 |
+
"special": false
|
| 53 |
+
},
|
| 54 |
+
"50258": {
|
| 55 |
+
"content": " ",
|
| 56 |
+
"lstrip": false,
|
| 57 |
+
"normalized": true,
|
| 58 |
+
"rstrip": false,
|
| 59 |
+
"single_word": false,
|
| 60 |
+
"special": false
|
| 61 |
+
},
|
| 62 |
+
"50259": {
|
| 63 |
+
"content": " ",
|
| 64 |
+
"lstrip": false,
|
| 65 |
+
"normalized": true,
|
| 66 |
+
"rstrip": false,
|
| 67 |
+
"single_word": false,
|
| 68 |
+
"special": false
|
| 69 |
+
},
|
| 70 |
+
"50260": {
|
| 71 |
+
"content": " ",
|
| 72 |
+
"lstrip": false,
|
| 73 |
+
"normalized": true,
|
| 74 |
+
"rstrip": false,
|
| 75 |
+
"single_word": false,
|
| 76 |
+
"special": false
|
| 77 |
+
},
|
| 78 |
+
"50261": {
|
| 79 |
+
"content": " ",
|
| 80 |
+
"lstrip": false,
|
| 81 |
+
"normalized": true,
|
| 82 |
+
"rstrip": false,
|
| 83 |
+
"single_word": false,
|
| 84 |
+
"special": false
|
| 85 |
+
},
|
| 86 |
+
"50262": {
|
| 87 |
+
"content": " ",
|
| 88 |
+
"lstrip": false,
|
| 89 |
+
"normalized": true,
|
| 90 |
+
"rstrip": false,
|
| 91 |
+
"single_word": false,
|
| 92 |
+
"special": false
|
| 93 |
+
},
|
| 94 |
+
"50263": {
|
| 95 |
+
"content": " ",
|
| 96 |
+
"lstrip": false,
|
| 97 |
+
"normalized": true,
|
| 98 |
+
"rstrip": false,
|
| 99 |
+
"single_word": false,
|
| 100 |
+
"special": false
|
| 101 |
+
},
|
| 102 |
+
"50264": {
|
| 103 |
+
"content": " ",
|
| 104 |
+
"lstrip": false,
|
| 105 |
+
"normalized": true,
|
| 106 |
+
"rstrip": false,
|
| 107 |
+
"single_word": false,
|
| 108 |
+
"special": false
|
| 109 |
+
},
|
| 110 |
+
"50265": {
|
| 111 |
+
"content": " ",
|
| 112 |
+
"lstrip": false,
|
| 113 |
+
"normalized": true,
|
| 114 |
+
"rstrip": false,
|
| 115 |
+
"single_word": false,
|
| 116 |
+
"special": false
|
| 117 |
+
},
|
| 118 |
+
"50266": {
|
| 119 |
+
"content": " ",
|
| 120 |
+
"lstrip": false,
|
| 121 |
+
"normalized": true,
|
| 122 |
+
"rstrip": false,
|
| 123 |
+
"single_word": false,
|
| 124 |
+
"special": false
|
| 125 |
+
},
|
| 126 |
+
"50267": {
|
| 127 |
+
"content": " ",
|
| 128 |
+
"lstrip": false,
|
| 129 |
+
"normalized": true,
|
| 130 |
+
"rstrip": false,
|
| 131 |
+
"single_word": false,
|
| 132 |
+
"special": false
|
| 133 |
+
},
|
| 134 |
+
"50268": {
|
| 135 |
+
"content": " ",
|
| 136 |
+
"lstrip": false,
|
| 137 |
+
"normalized": true,
|
| 138 |
+
"rstrip": false,
|
| 139 |
+
"single_word": false,
|
| 140 |
+
"special": false
|
| 141 |
+
},
|
| 142 |
+
"50269": {
|
| 143 |
+
"content": " ",
|
| 144 |
+
"lstrip": false,
|
| 145 |
+
"normalized": true,
|
| 146 |
+
"rstrip": false,
|
| 147 |
+
"single_word": false,
|
| 148 |
+
"special": false
|
| 149 |
+
},
|
| 150 |
+
"50270": {
|
| 151 |
+
"content": " ",
|
| 152 |
+
"lstrip": false,
|
| 153 |
+
"normalized": true,
|
| 154 |
+
"rstrip": false,
|
| 155 |
+
"single_word": false,
|
| 156 |
+
"special": false
|
| 157 |
+
},
|
| 158 |
+
"50271": {
|
| 159 |
+
"content": " ",
|
| 160 |
+
"lstrip": false,
|
| 161 |
+
"normalized": true,
|
| 162 |
+
"rstrip": false,
|
| 163 |
+
"single_word": false,
|
| 164 |
+
"special": false
|
| 165 |
+
},
|
| 166 |
+
"50272": {
|
| 167 |
+
"content": " ",
|
| 168 |
+
"lstrip": false,
|
| 169 |
+
"normalized": true,
|
| 170 |
+
"rstrip": false,
|
| 171 |
+
"single_word": false,
|
| 172 |
+
"special": false
|
| 173 |
+
},
|
| 174 |
+
"50273": {
|
| 175 |
+
"content": " ",
|
| 176 |
+
"lstrip": false,
|
| 177 |
+
"normalized": true,
|
| 178 |
+
"rstrip": false,
|
| 179 |
+
"single_word": false,
|
| 180 |
+
"special": false
|
| 181 |
+
},
|
| 182 |
+
"50274": {
|
| 183 |
+
"content": " ",
|
| 184 |
+
"lstrip": false,
|
| 185 |
+
"normalized": true,
|
| 186 |
+
"rstrip": false,
|
| 187 |
+
"single_word": false,
|
| 188 |
+
"special": false
|
| 189 |
+
},
|
| 190 |
+
"50275": {
|
| 191 |
+
"content": " ",
|
| 192 |
+
"lstrip": false,
|
| 193 |
+
"normalized": true,
|
| 194 |
+
"rstrip": false,
|
| 195 |
+
"single_word": false,
|
| 196 |
+
"special": false
|
| 197 |
+
},
|
| 198 |
+
"50276": {
|
| 199 |
+
"content": " ",
|
| 200 |
+
"lstrip": false,
|
| 201 |
+
"normalized": true,
|
| 202 |
+
"rstrip": false,
|
| 203 |
+
"single_word": false,
|
| 204 |
+
"special": false
|
| 205 |
+
},
|
| 206 |
+
"50277": {
|
| 207 |
+
"content": "|||EMAIL_ADDRESS|||",
|
| 208 |
+
"lstrip": false,
|
| 209 |
+
"normalized": true,
|
| 210 |
+
"rstrip": false,
|
| 211 |
+
"single_word": false,
|
| 212 |
+
"special": false
|
| 213 |
+
},
|
| 214 |
+
"50278": {
|
| 215 |
+
"content": "|||PHONE_NUMBER|||",
|
| 216 |
+
"lstrip": false,
|
| 217 |
+
"normalized": true,
|
| 218 |
+
"rstrip": false,
|
| 219 |
+
"single_word": false,
|
| 220 |
+
"special": false
|
| 221 |
+
},
|
| 222 |
+
"50279": {
|
| 223 |
+
"content": "|||IP_ADDRESS|||",
|
| 224 |
+
"lstrip": false,
|
| 225 |
+
"normalized": true,
|
| 226 |
+
"rstrip": false,
|
| 227 |
+
"single_word": false,
|
| 228 |
+
"special": true
|
| 229 |
+
},
|
| 230 |
+
"50280": {
|
| 231 |
+
"content": "<pad>",
|
| 232 |
+
"lstrip": false,
|
| 233 |
+
"normalized": false,
|
| 234 |
+
"rstrip": false,
|
| 235 |
+
"single_word": false,
|
| 236 |
+
"special": true
|
| 237 |
+
}
|
| 238 |
+
},
|
| 239 |
+
"bos_token": "|||IP_ADDRESS|||",
|
| 240 |
+
"clean_up_tokenization_spaces": false,
|
| 241 |
+
"eos_token": "|||IP_ADDRESS|||",
|
| 242 |
+
"extra_special_tokens": {},
|
| 243 |
+
"model_max_length": 1000000000000000019884624838656,
|
| 244 |
+
"pad_token": "<pad>",
|
| 245 |
+
"tokenizer_class": "GPTNeoXTokenizer",
|
| 246 |
+
"unk_token": null
|
| 247 |
+
}
|
pruned/knee0924/keep20.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
saved 1.76B params (461/1024 experts deleted) -> outputs/pruned/knee0924/keep20 (3.5 GB safetensors)
|
pruned/knee0924/keep25.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
saved 2.09B params (382/1024 experts deleted) -> outputs/pruned/knee0924/keep25 (4.2 GB safetensors)
|
pruned/knee0924/keep30.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
saved 2.40B params (317/1024 experts deleted) -> outputs/pruned/knee0924/keep30 (4.8 GB safetensors)
|
pruned/knee0924/keep50.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
saved 3.70B params (186/1024 experts deleted) -> outputs/pruned/knee0924/keep50 (7.4 GB safetensors)
|