Add files using upload-large-folder tool
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- healed/calib/glean_keep25_s1224.console.log +0 -0
- healed/calib/glean_keep25_s1224.eval.log +81 -0
- healed/calib/glean_keep25_s1224/train_log.jsonl +0 -0
- healed/calib/glean_keep25_s1224/wandb/run-20260717_210858-ypxqb0r8/files/config.yaml +246 -0
- healed/calib/glean_keep25_s1224/wandb/run-20260717_210858-ypxqb0r8/files/output.log +115 -0
- healed/calib/glean_keep25_s1224/wandb/run-20260717_210858-ypxqb0r8/logs/debug-core.log +20 -0
- healed/calib/glean_keep25_s1224/wandb/run-20260717_210858-ypxqb0r8/logs/debug.log +24 -0
- healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/files/output.log +59 -0
- healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/files/requirements.txt +129 -0
- healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/files/wandb-metadata.json +97 -0
- healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/logs/debug-core.log +7 -0
- healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/logs/debug-internal.log +335 -0
- healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/logs/debug.log +19 -0
- healed/calib/glean_keep50_s1224.console.log +0 -0
- healed/calib/glean_keep50_s1224.eval.log +81 -0
- healed/calib/glean_keep50_s1224/args.json +62 -0
- healed/calib/glean_keep75_s1224.console.log +0 -0
- healed/calib/glean_keep75_s1224.eval.log +54 -0
- healed/healing_breadth/glean_math_keep25_seed1224.console.log +120 -0
- healed/healing_breadth/glean_math_keep25_seed1224_long.console.log +240 -0
- healed/healing_breadth/glean_math_keep25_seed1224_long768.console.log +0 -0
- healed/healing_breadth/keep25_long.queue.log +1 -0
- healed/healing_breadth/reap_math_keep75_seed1224.console.log +113 -0
- healed/healing_breadth/traj_sweep.log +88 -0
- healed/healing_breadth/traj_sweep2.log +37 -0
- healed/keep50_warmup_fixed_s1224/args.json +72 -0
- healed/keep50_warmup_fixed_s1224/step0150/modeling_pruned_olmoe.py +66 -0
- healed/keep50_warmup_fixed_s1224/train_log.jsonl +150 -0
- healed/opd_warm_unleashed/args.json +71 -0
- healed/opd_warm_unleashed/step0450/chat_template.jinja +9 -0
- healed/opd_warm_unleashed/step0450/config.json +887 -0
- healed/opd_warm_unleashed/step0450/configuration_pruned_olmoe.py +27 -0
- healed/opd_warm_unleashed/step0450/generation_config.json +6 -0
- healed/opd_warm_unleashed/step0450/model.safetensors.index.json +0 -0
- healed/opd_warm_unleashed/step0450/modeling_pruned_olmoe.py +66 -0
- healed/opd_warm_unleashed/step0450/special_tokens_map.json +23 -0
- healed/opd_warm_unleashed/step0450/tokenizer.json +0 -0
- healed/opd_warm_unleashed/step0450/tokenizer_config.json +247 -0
- healed/opd_warm_unleashed/step0500/chat_template.jinja +9 -0
- healed/opd_warm_unleashed/step0500/config.json +887 -0
- healed/opd_warm_unleashed/step0500/configuration_pruned_olmoe.py +27 -0
- healed/opd_warm_unleashed/step0500/generation_config.json +6 -0
- healed/opd_warm_unleashed/step0500/model.safetensors.index.json +0 -0
- healed/opd_warm_unleashed/step0500/modeling_pruned_olmoe.py +66 -0
- healed/opd_warm_unleashed/step0500/special_tokens_map.json +23 -0
- healed/opd_warm_unleashed/step0500/tokenizer.json +0 -0
- healed/opd_warm_unleashed/step0500/tokenizer_config.json +247 -0
- healed/opd_warm_unleashed/train_log.jsonl +0 -0
- healed/opd_warm_unleashed/vllm_server.log +0 -0
- healed/opd_warm_unleashed/wandb/debug-internal.log +16 -0
healed/calib/glean_keep25_s1224.console.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/calib/glean_keep25_s1224.eval.log
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 1 |
38%|███▊ | 189/500 [00:00<00:00, 1882.41it/s]
|
| 2 |
77%|███████▋ | 386/500 [00:00<00:00, 1932.60it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 4 |
37%|███▋ | 185/500 [00:00<00:00, 1848.56it/s]
|
| 5 |
76%|███████▌ | 380/500 [00:00<00:00, 1903.49it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 6 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 7 |
37%|███▋ | 187/500 [00:00<00:00, 1862.95it/s]
|
| 8 |
76%|███████▋ | 382/500 [00:00<00:00, 1910.76it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T21:06:49-07:00 serving outputs/healed/calib/glean_keep25_s1224/step0100 on GPU 0 port 8430
|
| 2 |
+
2026-07-17T21:06:49-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T21:07:14-07:00 server up; chat pass [gsm8k_cot_zeroshot]
|
| 4 |
+
2026-07-17:21:07:15 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 5 |
+
2026-07-17:21:07:22 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot']
|
| 6 |
+
2026-07-17:21:07:23 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 7 |
+
2026-07-17:21:07:23 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding!
|
| 8 |
+
2026-07-17:21:07:23 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8430/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 9 |
+
2026-07-17:21:07:23 INFO [models.api_models:179] Using max length 2048 - 1
|
| 10 |
+
2026-07-17:21:07:23 INFO [models.api_models:200] Using tokenizer None
|
| 11 |
+
2026-07-17:21:07:25 INFO [evaluator_utils:446] Selected tasks:
|
| 12 |
+
2026-07-17:21:07:25 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 13 |
+
2026-07-17:21:07:25 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280}
|
| 14 |
+
2026-07-17:21:07:25 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 15 |
+
|
| 16 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 17 |
38%|███▊ | 189/500 [00:00<00:00, 1882.41it/s]
|
| 18 |
77%|███████▋ | 386/500 [00:00<00:00, 1932.60it/s]
|
| 19 |
+
2026-07-17:21:07:25 INFO [evaluator:585] Running generate_until requests
|
| 20 |
+
2026-07-17:21:07:25 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 21 |
+
|
| 22 |
+
2026-07-17:21:08:50 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 23 |
+
2026-07-17:21:08:50 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/calib/keep25_step100/student/*.jsonl
|
| 24 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8430/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({'max_gen_toks': 1280}), limit: 500.0, num_fewshot: None, batch_size: 1
|
| 25 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 26 |
+
|------------------|------:|----------------|-----:|-----------|---|----:|---|-----:|
|
| 27 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match|↑ |0.182|± |0.0173|
|
| 28 |
+
| | |strict-match | 0|exact_match|↑ |0.000|± |0.0000|
|
| 29 |
+
|
| 30 |
+
2026-07-17T21:08:51-07:00 lm_eval exit=0 -> outputs/evals/general_suite/calib/keep25_step100
|
| 31 |
+
2026-07-17T22:31:43-07:00 serving outputs/healed/calib/glean_keep25_s1224/step0200 on GPU 0 port 8430
|
| 32 |
+
2026-07-17T22:31:43-07:00 waiting for server /health ...
|
| 33 |
+
2026-07-17T22:32:08-07:00 server up; chat pass [gsm8k_cot_zeroshot]
|
| 34 |
+
2026-07-17:22:32:08 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 35 |
+
2026-07-17:22:32:15 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot']
|
| 36 |
+
2026-07-17:22:32:16 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 37 |
+
2026-07-17:22:32:16 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding!
|
| 38 |
+
2026-07-17:22:32:16 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8430/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 39 |
+
2026-07-17:22:32:16 INFO [models.api_models:179] Using max length 2048 - 1
|
| 40 |
+
2026-07-17:22:32:16 INFO [models.api_models:200] Using tokenizer None
|
| 41 |
+
2026-07-17:22:32:18 INFO [evaluator_utils:446] Selected tasks:
|
| 42 |
+
2026-07-17:22:32:18 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 43 |
+
2026-07-17:22:32:18 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280}
|
| 44 |
+
2026-07-17:22:32:18 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 45 |
+
|
| 46 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 47 |
37%|███▋ | 185/500 [00:00<00:00, 1848.56it/s]
|
| 48 |
76%|███████▌ | 380/500 [00:00<00:00, 1903.49it/s]
|
| 49 |
+
2026-07-17:22:32:18 INFO [evaluator:585] Running generate_until requests
|
| 50 |
+
2026-07-17:22:32:18 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 51 |
+
|
| 52 |
+
2026-07-17:22:33:31 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 53 |
+
2026-07-17:22:33:31 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/calib/keep25_step200/student/*.jsonl
|
| 54 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8430/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({'max_gen_toks': 1280}), limit: 500.0, num_fewshot: None, batch_size: 1
|
| 55 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 56 |
+
|------------------|------:|----------------|-----:|-----------|---|----:|---|-----:|
|
| 57 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match|↑ |0.238|± |0.0191|
|
| 58 |
+
| | |strict-match | 0|exact_match|↑ |0.000|± |0.0000|
|
| 59 |
+
|
| 60 |
+
2026-07-17T22:33:32-07:00 lm_eval exit=0 -> outputs/evals/general_suite/calib/keep25_step200
|
| 61 |
+
2026-07-17T23:55:53-07:00 serving outputs/healed/calib/glean_keep25_s1224/step0300 on GPU 0 port 8430
|
| 62 |
+
2026-07-17T23:55:53-07:00 waiting for server /health ...
|
| 63 |
+
2026-07-17T23:56:18-07:00 server up; chat pass [gsm8k_cot_zeroshot]
|
| 64 |
+
2026-07-17:23:56:18 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 65 |
+
2026-07-17:23:56:25 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot']
|
| 66 |
+
2026-07-17:23:56:26 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 67 |
+
2026-07-17:23:56:26 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding!
|
| 68 |
+
2026-07-17:23:56:26 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8430/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 69 |
+
2026-07-17:23:56:26 INFO [models.api_models:179] Using max length 2048 - 1
|
| 70 |
+
2026-07-17:23:56:26 INFO [models.api_models:200] Using tokenizer None
|
| 71 |
+
2026-07-17:23:56:28 INFO [evaluator_utils:446] Selected tasks:
|
| 72 |
+
2026-07-17:23:56:28 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 73 |
+
2026-07-17:23:56:28 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280}
|
| 74 |
+
2026-07-17:23:56:28 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 75 |
+
|
| 76 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 77 |
37%|███▋ | 187/500 [00:00<00:00, 1862.95it/s]
|
| 78 |
76%|███████▋ | 382/500 [00:00<00:00, 1910.76it/s]
|
| 79 |
+
2026-07-17:23:56:28 INFO [evaluator:585] Running generate_until requests
|
| 80 |
+
2026-07-17:23:56:28 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 81 |
+
|
| 82 |
+
2026-07-17:23:57:31 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 83 |
+
2026-07-17:23:57:31 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/calib/keep25_step300/student/*.jsonl
|
| 84 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8430/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({'max_gen_toks': 1280}), limit: 500.0, num_fewshot: None, batch_size: 1
|
| 85 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 86 |
+
|------------------|------:|----------------|-----:|-----------|---|----:|---|-----:|
|
| 87 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match|↑ |0.244|± |0.0192|
|
| 88 |
+
| | |strict-match | 0|exact_match|↑ |0.004|± |0.0028|
|
| 89 |
+
|
| 90 |
+
2026-07-17T23:57:32-07:00 lm_eval exit=0 -> outputs/evals/general_suite/calib/keep25_step300
|
healed/calib/glean_keep25_s1224/train_log.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/calib/glean_keep25_s1224/wandb/run-20260717_210858-ypxqb0r8/files/config.yaml
ADDED
|
@@ -0,0 +1,246 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
_wandb:
|
| 2 |
+
value:
|
| 3 |
+
cli_version: 0.28.0
|
| 4 |
+
e:
|
| 5 |
+
toks52dbikg8amgtxadijyojntrgsti4:
|
| 6 |
+
args:
|
| 7 |
+
- --training-mode
|
| 8 |
+
- off-policy
|
| 9 |
+
- --kl-direction
|
| 10 |
+
- forward
|
| 11 |
+
- --topk-targets
|
| 12 |
+
- outputs/teacher_trajectories/dolci_combined_top128
|
| 13 |
+
- --off-policy-frames
|
| 14 |
+
- chat
|
| 15 |
+
- --student-device
|
| 16 |
+
- cuda:0
|
| 17 |
+
- --student
|
| 18 |
+
- outputs/pruned/glean-0125inst-general-keep25
|
| 19 |
+
- --resume-from
|
| 20 |
+
- outputs/healed/calib/glean_keep25_s1224/step0100
|
| 21 |
+
- --lr
|
| 22 |
+
- "3e-5"
|
| 23 |
+
- --optimizer
|
| 24 |
+
- adamw8bit
|
| 25 |
+
- --epochs
|
| 26 |
+
- "9"
|
| 27 |
+
- --sweep
|
| 28 |
+
- "200"
|
| 29 |
+
- --micro-batch
|
| 30 |
+
- "3"
|
| 31 |
+
- --loss-tokens-per-step
|
| 32 |
+
- "120000"
|
| 33 |
+
- --gsm8k-every
|
| 34 |
+
- "0"
|
| 35 |
+
- --save-every
|
| 36 |
+
- "200"
|
| 37 |
+
- --seed
|
| 38 |
+
- "1224"
|
| 39 |
+
- --out-dir
|
| 40 |
+
- outputs/healed/calib/glean_keep25_s1224
|
| 41 |
+
- --wandb
|
| 42 |
+
- --wandb-mode
|
| 43 |
+
- online
|
| 44 |
+
- --wandb-project
|
| 45 |
+
- glean-calib
|
| 46 |
+
- --wandb-run-name
|
| 47 |
+
- calib_keep25_200
|
| 48 |
+
codePath: scripts/11_distill_on_policy.py
|
| 49 |
+
codePathLocal: scripts/11_distill_on_policy.py
|
| 50 |
+
cpu_count: 24
|
| 51 |
+
cpu_count_logical: 48
|
| 52 |
+
cudaVersion: "13.0"
|
| 53 |
+
disk:
|
| 54 |
+
/:
|
| 55 |
+
total: "1958315118592"
|
| 56 |
+
used: "1632623116288"
|
| 57 |
+
email: hbfreed@protonmail.com
|
| 58 |
+
executable: /home/henry/Documents/PythonProjects/variable-reap/.venv/bin/python3
|
| 59 |
+
git:
|
| 60 |
+
commit: 247c7f0113d7a058b1db8044deddfd4456cc998d
|
| 61 |
+
remote: https://github.com/hbfreed/variable-reap.git
|
| 62 |
+
gpu: NVIDIA GeForce RTX 3090
|
| 63 |
+
gpu_count: 3
|
| 64 |
+
gpu_nvidia:
|
| 65 |
+
- architecture: Ampere
|
| 66 |
+
cudaCores: 10496
|
| 67 |
+
memoryTotal: "25769803776"
|
| 68 |
+
name: NVIDIA GeForce RTX 3090
|
| 69 |
+
uuid: GPU-8ca70870-ddf2-d274-bcc5-182c2075bced
|
| 70 |
+
- architecture: Ampere
|
| 71 |
+
cudaCores: 10496
|
| 72 |
+
memoryTotal: "25769803776"
|
| 73 |
+
name: NVIDIA GeForce RTX 3090
|
| 74 |
+
uuid: GPU-a6acf07f-31f5-618f-a5d0-c0017e7e2e27
|
| 75 |
+
- architecture: Ampere
|
| 76 |
+
cudaCores: 10496
|
| 77 |
+
memoryTotal: "25769803776"
|
| 78 |
+
name: NVIDIA GeForce RTX 3090
|
| 79 |
+
uuid: GPU-864c54df-0130-7780-e271-8a5551d1733f
|
| 80 |
+
host: pop-os
|
| 81 |
+
memory:
|
| 82 |
+
total: "134900756480"
|
| 83 |
+
os: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35
|
| 84 |
+
program: /home/henry/Documents/PythonProjects/variable-reap/scripts/11_distill_on_policy.py
|
| 85 |
+
python: CPython 3.12.12
|
| 86 |
+
root: outputs/healed/calib/glean_keep25_s1224
|
| 87 |
+
startedAt: "2026-07-18T04:08:58.716590Z"
|
| 88 |
+
writerId: toks52dbikg8amgtxadijyojntrgsti4
|
| 89 |
+
m:
|
| 90 |
+
- "1": step
|
| 91 |
+
"6":
|
| 92 |
+
- 3
|
| 93 |
+
"7": []
|
| 94 |
+
- "2": '*'
|
| 95 |
+
"5": 1
|
| 96 |
+
"6":
|
| 97 |
+
- 1
|
| 98 |
+
"7": []
|
| 99 |
+
python_version: 3.12.12
|
| 100 |
+
t:
|
| 101 |
+
"1":
|
| 102 |
+
- 1
|
| 103 |
+
- 5
|
| 104 |
+
- 11
|
| 105 |
+
- 49
|
| 106 |
+
- 51
|
| 107 |
+
- 53
|
| 108 |
+
- 71
|
| 109 |
+
"2":
|
| 110 |
+
- 1
|
| 111 |
+
- 5
|
| 112 |
+
- 11
|
| 113 |
+
- 49
|
| 114 |
+
- 51
|
| 115 |
+
- 53
|
| 116 |
+
- 71
|
| 117 |
+
"3":
|
| 118 |
+
- 2
|
| 119 |
+
- 7
|
| 120 |
+
- 13
|
| 121 |
+
- 16
|
| 122 |
+
"4": 3.12.12
|
| 123 |
+
"5": 0.28.0
|
| 124 |
+
"6": 4.57.6
|
| 125 |
+
"12": 0.28.0
|
| 126 |
+
"13": linux-x86_64
|
| 127 |
+
dataset:
|
| 128 |
+
value: allenai/RLVR-MATH
|
| 129 |
+
dataset_sources:
|
| 130 |
+
value: null
|
| 131 |
+
debug:
|
| 132 |
+
value: false
|
| 133 |
+
epochs:
|
| 134 |
+
value: 9
|
| 135 |
+
eval_every:
|
| 136 |
+
value: 10
|
| 137 |
+
group_size:
|
| 138 |
+
value: 1
|
| 139 |
+
gsm8k_batch:
|
| 140 |
+
value: 16
|
| 141 |
+
gsm8k_every:
|
| 142 |
+
value: 0
|
| 143 |
+
gsm8k_frames:
|
| 144 |
+
value: chat
|
| 145 |
+
gsm8k_max_new_tokens:
|
| 146 |
+
value: 512
|
| 147 |
+
gsm8k_n:
|
| 148 |
+
value: 256
|
| 149 |
+
kl_direction:
|
| 150 |
+
value: forward
|
| 151 |
+
loss_tokens_per_step:
|
| 152 |
+
value: 120000
|
| 153 |
+
lr:
|
| 154 |
+
value: 3e-05
|
| 155 |
+
max_difficulty:
|
| 156 |
+
value: null
|
| 157 |
+
max_grad_norm:
|
| 158 |
+
value: 1
|
| 159 |
+
max_loss_tokens:
|
| 160 |
+
value: null
|
| 161 |
+
max_new_tokens:
|
| 162 |
+
value: 256
|
| 163 |
+
max_prompt_len:
|
| 164 |
+
value: 1024
|
| 165 |
+
micro_batch:
|
| 166 |
+
value: 3
|
| 167 |
+
no_grad_checkpointing:
|
| 168 |
+
value: false
|
| 169 |
+
no_teacher_overlap:
|
| 170 |
+
value: false
|
| 171 |
+
no_wandb_sync:
|
| 172 |
+
value: false
|
| 173 |
+
off_policy_frames:
|
| 174 |
+
value: chat
|
| 175 |
+
off_policy_max_seq_len:
|
| 176 |
+
value: 2048
|
| 177 |
+
optimizer:
|
| 178 |
+
value: adamw8bit
|
| 179 |
+
out_dir:
|
| 180 |
+
value: outputs/healed/calib/glean_keep25_s1224
|
| 181 |
+
prompts_per_step:
|
| 182 |
+
value: 256
|
| 183 |
+
resolved_kl_direction:
|
| 184 |
+
value: forward
|
| 185 |
+
resume_from:
|
| 186 |
+
value: outputs/healed/calib/glean_keep25_s1224/step0100
|
| 187 |
+
rollout_batch:
|
| 188 |
+
value: 64
|
| 189 |
+
rollout_engine:
|
| 190 |
+
value: hf
|
| 191 |
+
save_every:
|
| 192 |
+
value: 200
|
| 193 |
+
seed:
|
| 194 |
+
value: 1224
|
| 195 |
+
start_step:
|
| 196 |
+
value: 0
|
| 197 |
+
student:
|
| 198 |
+
value: outputs/pruned/glean-0125inst-general-keep25
|
| 199 |
+
student_device:
|
| 200 |
+
value: cuda:0
|
| 201 |
+
sweep:
|
| 202 |
+
value: 200
|
| 203 |
+
sync_checkpoints:
|
| 204 |
+
value: false
|
| 205 |
+
teacher:
|
| 206 |
+
value: allenai/OLMoE-1B-7B-0125-Instruct
|
| 207 |
+
teacher_device:
|
| 208 |
+
value: cuda:0
|
| 209 |
+
topk_targets:
|
| 210 |
+
value: outputs/teacher_trajectories/dolci_combined_top128
|
| 211 |
+
training_mode:
|
| 212 |
+
value: off-policy
|
| 213 |
+
trajectories:
|
| 214 |
+
value: outputs/teacher_trajectories/dolci_math_curated.jsonl
|
| 215 |
+
trajectory_dataset:
|
| 216 |
+
value: allenai/Dolci-Instruct-RL
|
| 217 |
+
vllm_gpu:
|
| 218 |
+
value: null
|
| 219 |
+
vllm_gpu_mem_util:
|
| 220 |
+
value: 0.85
|
| 221 |
+
vllm_live_dir:
|
| 222 |
+
value: null
|
| 223 |
+
vllm_port:
|
| 224 |
+
value: 8377
|
| 225 |
+
vllm_refresh_every:
|
| 226 |
+
value: 5
|
| 227 |
+
vllm_refresh_mode:
|
| 228 |
+
value: reload
|
| 229 |
+
vllm_serve_bin:
|
| 230 |
+
value: vllm-plugin/.venv/bin/python
|
| 231 |
+
wandb:
|
| 232 |
+
value: true
|
| 233 |
+
wandb_mode:
|
| 234 |
+
value: online
|
| 235 |
+
wandb_project:
|
| 236 |
+
value: glean-calib
|
| 237 |
+
wandb_resume:
|
| 238 |
+
value: null
|
| 239 |
+
wandb_run_id:
|
| 240 |
+
value: null
|
| 241 |
+
wandb_run_name:
|
| 242 |
+
value: calib_keep25_200
|
| 243 |
+
warmup_steps:
|
| 244 |
+
value: 10
|
| 245 |
+
weight_decay:
|
| 246 |
+
value: 0.1
|
healed/calib/glean_keep25_s1224/wandb/run-20260717_210858-ypxqb0r8/files/output.log
ADDED
|
@@ -0,0 +1,115 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
resumed student weights from outputs/healed/calib/glean_keep25_s1224/step0100 (fresh optimizer, step counter at 0)
|
| 2 |
+
58360 cached top-128 chat trajectories / 22,295,631 unique tokens | 185 steps/epoch | 200 total steps | student params 2.09B | teacher overlap=False
|
| 3 |
+
restored optimizer/scheduler state from step 100; rebuilt 228 paged buffers
|
| 4 |
+
{"step": 101, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6404966282300651, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.6953125, "lr": 3e-05, "finish_rate": 0.994, "comp_len": 343.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 349}, "mem_gb": 9.61}
|
| 5 |
+
{"step": 102, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6106020681867376, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.71484375, "lr": 3e-05, "finish_rate": 0.968, "comp_len": 384.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 312}, "mem_gb": 9.73}
|
| 6 |
+
{"step": 103, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.22530150232814875, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.4921875, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.0, "frames": {"chat": 229}, "mem_gb": 10.0}
|
| 7 |
+
{"step": 104, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5713284558854997, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.66796875, "lr": 3e-05, "finish_rate": 0.938, "comp_len": 416.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 288}, "mem_gb": 10.0}
|
| 8 |
+
{"step": 105, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7972744106804331, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.8828125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 341.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.7, "frames": {"chat": 351}, "mem_gb": 9.66}
|
| 9 |
+
{"step": 106, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3168164167162031, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.812, "comp_len": 538.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.9, "frames": {"chat": 223}, "mem_gb": 10.02}
|
| 10 |
+
{"step": 107, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.15695743769599746, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.455078125, "lr": 3e-05, "finish_rate": 0.793, "comp_len": 563.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 36.4, "frames": {"chat": 213}, "mem_gb": 9.99}
|
| 11 |
+
{"step": 108, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7303095779831211, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.77734375, "lr": 3e-05, "finish_rate": 0.995, "comp_len": 315.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.1, "frames": {"chat": 380}, "mem_gb": 9.84}
|
| 12 |
+
{"step": 109, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4183019400515283, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.9, "frames": {"chat": 271}, "mem_gb": 10.02}
|
| 13 |
+
{"step": 110, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7819241975242893, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.73828125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 352.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 340}, "mem_gb": 9.91}
|
| 14 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 15 |
+
[eval step 110] sample: 'To determine the rank of a matrix, we need to find the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1 & '
|
| 16 |
+
{"step": 111, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8300092002927947, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.8125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 327.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.0, "frames": {"chat": 366}, "mem_gb": 9.61}
|
| 17 |
+
{"step": 112, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2994033165920526, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.5, "lr": 3e-05, "finish_rate": 0.854, "comp_len": 515.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.5, "frames": {"chat": 233}, "mem_gb": 10.0}
|
| 18 |
+
{"step": 113, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5316771011674156, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.968, "comp_len": 384.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.1, "frames": {"chat": 312}, "mem_gb": 9.86}
|
| 19 |
+
{"step": 114, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7487247935026263, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.7734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 363}, "mem_gb": 9.58}
|
| 20 |
+
{"step": 115, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7562611924018711, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.7578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 371.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.1, "frames": {"chat": 323}, "mem_gb": 9.85}
|
| 21 |
+
{"step": 116, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8215424215157827, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.80859375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 346.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 346}, "mem_gb": 9.59}
|
| 22 |
+
{"step": 117, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7370786963778237, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 1.0234375, "lr": 3e-05, "finish_rate": 0.975, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.3, "frames": {"chat": 318}, "mem_gb": 9.74}
|
| 23 |
+
{"step": 118, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5495684528866162, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.64453125, "lr": 3e-05, "finish_rate": 0.935, "comp_len": 436.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 275}, "mem_gb": 9.93}
|
| 24 |
+
{"step": 119, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6848975222120062, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.76953125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 347.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.9, "frames": {"chat": 345}, "mem_gb": 9.67}
|
| 25 |
+
{"step": 120, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6866140180771549, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.80078125, "lr": 3e-05, "finish_rate": 0.992, "comp_len": 318.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.8, "frames": {"chat": 377}, "mem_gb": 9.82}
|
| 26 |
+
[eval step 120] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1 & '
|
| 27 |
+
{"step": 121, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5571080892120799, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.66796875, "lr": 3e-05, "finish_rate": 0.956, "comp_len": 408.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 294}, "mem_gb": 10.04}
|
| 28 |
+
{"step": 122, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7723195617940277, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.7578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 355.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.7, "frames": {"chat": 338}, "mem_gb": 9.73}
|
| 29 |
+
{"step": 123, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7758043260567511, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.76953125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 340.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.1, "frames": {"chat": 352}, "mem_gb": 9.63}
|
| 30 |
+
{"step": 124, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7300908756038795, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.765625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 345.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.2, "frames": {"chat": 347}, "mem_gb": 9.68}
|
| 31 |
+
{"step": 125, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.44740089606537176, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.65625, "lr": 3e-05, "finish_rate": 0.968, "comp_len": 387.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 310}, "mem_gb": 9.66}
|
| 32 |
+
{"step": 126, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8197008320604762, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.84375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.7, "frames": {"chat": 358}, "mem_gb": 9.69}
|
| 33 |
+
{"step": 127, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.47266409537059567, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.68359375, "lr": 3e-05, "finish_rate": 0.947, "comp_len": 376.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 319}, "mem_gb": 9.97}
|
| 34 |
+
{"step": 128, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.31353171376734973, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 446.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.2, "frames": {"chat": 269}, "mem_gb": 10.0}
|
| 35 |
+
{"step": 129, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6114998117436965, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.796875, "lr": 3e-05, "finish_rate": 0.974, "comp_len": 387.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.0, "frames": {"chat": 310}, "mem_gb": 9.85}
|
| 36 |
+
{"step": 130, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.39959967873397595, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.62890625, "lr": 3e-05, "finish_rate": 0.834, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.2, "frames": {"chat": 259}, "mem_gb": 10.03}
|
| 37 |
+
[eval step 130] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly independe'
|
| 38 |
+
{"step": 131, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3943806648887073, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 452.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.2, "frames": {"chat": 265}, "mem_gb": 10.0}
|
| 39 |
+
{"step": 132, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7735084446456283, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.82421875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 356.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.7, "frames": {"chat": 337}, "mem_gb": 9.55}
|
| 40 |
+
{"step": 133, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.43104476007341097, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.625, "lr": 3e-05, "finish_rate": 0.972, "comp_len": 413.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.8, "frames": {"chat": 290}, "mem_gb": 9.81}
|
| 41 |
+
{"step": 134, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6807688963243738, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.7578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 343.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 349}, "mem_gb": 9.75}
|
| 42 |
+
{"step": 135, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.43767833087344965, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.609375, "lr": 3e-05, "finish_rate": 0.928, "comp_len": 412.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.8, "frames": {"chat": 291}, "mem_gb": 9.84}
|
| 43 |
+
{"step": 136, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6971611228440577, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.7734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.9, "frames": {"chat": 342}, "mem_gb": 9.71}
|
| 44 |
+
{"step": 137, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8007253739355132, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.77734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 355.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.8, "frames": {"chat": 338}, "mem_gb": 9.66}
|
| 45 |
+
{"step": 138, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7883249476203074, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.78515625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 337.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.6, "frames": {"chat": 356}, "mem_gb": 9.62}
|
| 46 |
+
{"step": 139, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.36883119474363824, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.63671875, "lr": 3e-05, "finish_rate": 0.86, "comp_len": 452.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.2, "frames": {"chat": 265}, "mem_gb": 10.0}
|
| 47 |
+
{"step": 140, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6382841675106437, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.765625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 353}, "mem_gb": 9.71}
|
| 48 |
+
[eval step 140] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1'
|
| 49 |
+
{"step": 141, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6029005106360341, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.703125, "lr": 3e-05, "finish_rate": 0.949, "comp_len": 379.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.6, "frames": {"chat": 316}, "mem_gb": 9.9}
|
| 50 |
+
{"step": 142, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5026192011914216, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.66796875, "lr": 3e-05, "finish_rate": 0.945, "comp_len": 415.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 289}, "mem_gb": 10.05}
|
| 51 |
+
{"step": 143, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5401858362942934, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.8046875, "lr": 3e-05, "finish_rate": 0.982, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.9, "frames": {"chat": 331}, "mem_gb": 9.69}
|
| 52 |
+
{"step": 144, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5338050864121566, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.69921875, "lr": 3e-05, "finish_rate": 0.972, "comp_len": 375.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.6, "frames": {"chat": 320}, "mem_gb": 10.0}
|
| 53 |
+
{"step": 145, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.48790957024022935, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.71875, "lr": 3e-05, "finish_rate": 0.941, "comp_len": 441.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.0, "frames": {"chat": 272}, "mem_gb": 9.74}
|
| 54 |
+
{"step": 146, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8456092888927087, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.83984375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 326.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.8, "frames": {"chat": 368}, "mem_gb": 9.66}
|
| 55 |
+
{"step": 147, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6982750026626512, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.76953125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 365.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.0, "frames": {"chat": 328}, "mem_gb": 9.68}
|
| 56 |
+
{"step": 148, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6740354536671191, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.75390625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 368.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.7, "frames": {"chat": 326}, "mem_gb": 9.9}
|
| 57 |
+
{"step": 149, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7793807467692221, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.78125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 329.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.5, "frames": {"chat": 364}, "mem_gb": 9.84}
|
| 58 |
+
{"step": 150, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4921399115661159, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.97, "comp_len": 405.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.5, "frames": {"chat": 296}, "mem_gb": 9.88}
|
| 59 |
+
[eval step 150] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's start by writing the given 4x4 matrix:\n\n\\[\n\\begin{bmatrix}\n12 & "
|
| 60 |
+
{"step": 151, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.751563901235722, "tokens": 120000, "cumulative_loss_tokens": 18120000, "grad_norm": 0.76953125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 317.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.7, "frames": {"chat": 378}, "mem_gb": 9.74}
|
| 61 |
+
{"step": 152, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8484429204796752, "tokens": 120000, "cumulative_loss_tokens": 18240000, "grad_norm": 0.8515625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 326.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.0, "frames": {"chat": 368}, "mem_gb": 9.83}
|
| 62 |
+
{"step": 153, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.385568957734108, "tokens": 120000, "cumulative_loss_tokens": 18360000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.898, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.8, "frames": {"chat": 256}, "mem_gb": 9.94}
|
| 63 |
+
{"step": 154, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.42377624020079774, "tokens": 120000, "cumulative_loss_tokens": 18480000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.93, "comp_len": 381.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 315}, "mem_gb": 9.89}
|
| 64 |
+
{"step": 155, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.40477768267107506, "tokens": 120000, "cumulative_loss_tokens": 18600000, "grad_norm": 0.58984375, "lr": 3e-05, "finish_rate": 0.974, "comp_len": 384.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.4, "frames": {"chat": 312}, "mem_gb": 9.81}
|
| 65 |
+
{"step": 156, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5997266300438593, "tokens": 120000, "cumulative_loss_tokens": 18720000, "grad_norm": 0.69921875, "lr": 3e-05, "finish_rate": 0.97, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.5, "frames": {"chat": 330}, "mem_gb": 10.0}
|
| 66 |
+
{"step": 157, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5225513759051139, "tokens": 120000, "cumulative_loss_tokens": 18840000, "grad_norm": 0.6796875, "lr": 3e-05, "finish_rate": 0.954, "comp_len": 367.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.3, "frames": {"chat": 327}, "mem_gb": 10.0}
|
| 67 |
+
{"step": 158, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6792106815474729, "tokens": 120000, "cumulative_loss_tokens": 18960000, "grad_norm": 0.734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 341.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.4, "frames": {"chat": 351}, "mem_gb": 9.66}
|
| 68 |
+
{"step": 159, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4116748679873844, "tokens": 120000, "cumulative_loss_tokens": 19080000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.973, "comp_len": 357.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 336}, "mem_gb": 9.71}
|
| 69 |
+
{"step": 160, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7196924281466132, "tokens": 120000, "cumulative_loss_tokens": 19200000, "grad_norm": 0.79296875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 332.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.5, "frames": {"chat": 361}, "mem_gb": 9.68}
|
| 70 |
+
[eval step 160] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1'
|
| 71 |
+
{"step": 161, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7385822186128547, "tokens": 120000, "cumulative_loss_tokens": 19320000, "grad_norm": 0.75390625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 328.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.3, "frames": {"chat": 365}, "mem_gb": 9.63}
|
| 72 |
+
{"step": 162, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6650009809102863, "tokens": 120000, "cumulative_loss_tokens": 19440000, "grad_norm": 0.72265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 364.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 329}, "mem_gb": 9.68}
|
| 73 |
+
{"step": 163, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4450254848840957, "tokens": 120000, "cumulative_loss_tokens": 19560000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.931, "comp_len": 416.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 288}, "mem_gb": 9.89}
|
| 74 |
+
{"step": 164, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3387074120278781, "tokens": 120000, "cumulative_loss_tokens": 19680000, "grad_norm": 0.59765625, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 451.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.9, "frames": {"chat": 266}, "mem_gb": 10.04}
|
| 75 |
+
{"step": 165, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.622777245139579, "tokens": 120000, "cumulative_loss_tokens": 19800000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 372.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.9, "frames": {"chat": 322}, "mem_gb": 9.96}
|
| 76 |
+
{"step": 166, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6013612735873709, "tokens": 120000, "cumulative_loss_tokens": 19920000, "grad_norm": 0.6875, "lr": 3e-05, "finish_rate": 0.989, "comp_len": 334.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.0, "frames": {"chat": 359}, "mem_gb": 9.66}
|
| 77 |
+
{"step": 167, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.5200983359214539, "tokens": 120000, "cumulative_loss_tokens": 20040000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.965, "comp_len": 385.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.3, "frames": {"chat": 311}, "mem_gb": 9.78}
|
| 78 |
+
{"step": 168, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4737990202021475, "tokens": 120000, "cumulative_loss_tokens": 20160000, "grad_norm": 0.60546875, "lr": 3e-05, "finish_rate": 0.95, "comp_len": 396.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.8, "frames": {"chat": 303}, "mem_gb": 9.84}
|
| 79 |
+
{"step": 169, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7251967468782018, "tokens": 120000, "cumulative_loss_tokens": 20280000, "grad_norm": 0.78125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 351.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 55.9, "frames": {"chat": 341}, "mem_gb": 9.85}
|
| 80 |
+
{"step": 170, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.18425790825231622, "tokens": 120000, "cumulative_loss_tokens": 20400000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.843, "comp_len": 524.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.5, "frames": {"chat": 229}, "mem_gb": 9.91}
|
| 81 |
+
[eval step 170] sample: "To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's start by writing the given 4x4 matrix:\n\n\\[\n\\begin{bmatrix}\n12 & "
|
| 82 |
+
{"step": 171, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4877417909882963, "tokens": 120000, "cumulative_loss_tokens": 20520000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.964, "comp_len": 389.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 308}, "mem_gb": 10.06}
|
| 83 |
+
{"step": 172, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6497918719161302, "tokens": 120000, "cumulative_loss_tokens": 20640000, "grad_norm": 0.734375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 331.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.7, "frames": {"chat": 362}, "mem_gb": 9.56}
|
| 84 |
+
{"step": 173, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6920267764383927, "tokens": 120000, "cumulative_loss_tokens": 20760000, "grad_norm": 0.7265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 315.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 58.4, "frames": {"chat": 380}, "mem_gb": 9.81}
|
| 85 |
+
{"step": 174, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6489797173403203, "tokens": 120000, "cumulative_loss_tokens": 20880000, "grad_norm": 0.7578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 336.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 59.3, "frames": {"chat": 357}, "mem_gb": 9.59}
|
| 86 |
+
{"step": 175, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3671841130634149, "tokens": 120000, "cumulative_loss_tokens": 21000000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.8, "frames": {"chat": 254}, "mem_gb": 10.06}
|
| 87 |
+
{"step": 176, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3038933877159531, "tokens": 120000, "cumulative_loss_tokens": 21120000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.874, "comp_len": 459.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.3, "frames": {"chat": 261}, "mem_gb": 10.03}
|
| 88 |
+
{"step": 177, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6068330939687168, "tokens": 120000, "cumulative_loss_tokens": 21240000, "grad_norm": 0.7109375, "lr": 3e-05, "finish_rate": 0.982, "comp_len": 431.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.2, "frames": {"chat": 278}, "mem_gb": 9.82}
|
| 89 |
+
{"step": 178, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.6536359203963541, "tokens": 120000, "cumulative_loss_tokens": 21360000, "grad_norm": 0.67578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 360.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 333}, "mem_gb": 9.96}
|
| 90 |
+
{"step": 179, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.3293319945723439, "tokens": 120000, "cumulative_loss_tokens": 21480000, "grad_norm": 0.53515625, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.9, "frames": {"chat": 250}, "mem_gb": 9.92}
|
| 91 |
+
{"step": 180, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7798017734301587, "tokens": 120000, "cumulative_loss_tokens": 21600000, "grad_norm": 0.8671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 336.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.4, "frames": {"chat": 357}, "mem_gb": 9.7}
|
| 92 |
+
[eval step 180] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1'
|
| 93 |
+
{"step": 181, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.8047170679829704, "tokens": 120000, "cumulative_loss_tokens": 21720000, "grad_norm": 0.7421875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.5, "frames": {"chat": 330}, "mem_gb": 9.67}
|
| 94 |
+
{"step": 182, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.7163969376386454, "tokens": 120000, "cumulative_loss_tokens": 21840000, "grad_norm": 0.76171875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 341.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.8, "frames": {"chat": 351}, "mem_gb": 9.71}
|
| 95 |
+
{"step": 183, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.4812941745320335, "tokens": 120000, "cumulative_loss_tokens": 21960000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 413.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.0, "frames": {"chat": 290}, "mem_gb": 9.84}
|
| 96 |
+
{"step": 184, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.567710873114597, "tokens": 120000, "cumulative_loss_tokens": 22080000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.969, "comp_len": 372.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.5, "frames": {"chat": 322}, "mem_gb": 9.78}
|
| 97 |
+
{"step": 185, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.32963835290049515, "tokens": 120000, "cumulative_loss_tokens": 22200000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.827, "comp_len": 451.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.1, "frames": {"chat": 266}, "mem_gb": 10.01}
|
| 98 |
+
{"step": 186, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.6460849656254053, "tokens": 120000, "cumulative_loss_tokens": 22320000, "grad_norm": 0.92578125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 331}, "mem_gb": 9.86}
|
| 99 |
+
{"step": 187, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5878346645558874, "tokens": 120000, "cumulative_loss_tokens": 22440000, "grad_norm": 0.74609375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 350.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.1, "frames": {"chat": 342}, "mem_gb": 9.62}
|
| 100 |
+
{"step": 188, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.7007336048837751, "tokens": 120000, "cumulative_loss_tokens": 22560000, "grad_norm": 0.73046875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.9, "frames": {"chat": 358}, "mem_gb": 9.66}
|
| 101 |
+
{"step": 189, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5796603640264522, "tokens": 120000, "cumulative_loss_tokens": 22680000, "grad_norm": 0.7890625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 348.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.6, "frames": {"chat": 344}, "mem_gb": 9.75}
|
| 102 |
+
{"step": 190, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.37503966700950014, "tokens": 120000, "cumulative_loss_tokens": 22800000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 449.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.4, "frames": {"chat": 267}, "mem_gb": 10.06}
|
| 103 |
+
[eval step 190] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly independe'
|
| 104 |
+
{"step": 191, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.16379121993631124, "tokens": 120000, "cumulative_loss_tokens": 22920000, "grad_norm": 0.46875, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.6, "frames": {"chat": 248}, "mem_gb": 9.89}
|
| 105 |
+
{"step": 192, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5387236011125147, "tokens": 120000, "cumulative_loss_tokens": 23040000, "grad_norm": 0.62890625, "lr": 3e-05, "finish_rate": 0.979, "comp_len": 364.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.5, "frames": {"chat": 329}, "mem_gb": 9.89}
|
| 106 |
+
{"step": 193, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5913820276208221, "tokens": 120000, "cumulative_loss_tokens": 23160000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 354.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.5, "frames": {"chat": 339}, "mem_gb": 9.85}
|
| 107 |
+
{"step": 194, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.7116979550040637, "tokens": 120000, "cumulative_loss_tokens": 23280000, "grad_norm": 0.7265625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.4, "frames": {"chat": 354}, "mem_gb": 9.84}
|
| 108 |
+
{"step": 195, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3713194830263654, "tokens": 120000, "cumulative_loss_tokens": 23400000, "grad_norm": 0.58203125, "lr": 3e-05, "finish_rate": 0.943, "comp_len": 430.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.1, "frames": {"chat": 279}, "mem_gb": 9.98}
|
| 109 |
+
{"step": 196, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4681690845252946, "tokens": 120000, "cumulative_loss_tokens": 23520000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 377.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.0, "frames": {"chat": 318}, "mem_gb": 9.91}
|
| 110 |
+
{"step": 197, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3116183115092417, "tokens": 120000, "cumulative_loss_tokens": 23640000, "grad_norm": 0.54296875, "lr": 3e-05, "finish_rate": 0.875, "comp_len": 452.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 265}, "mem_gb": 9.99}
|
| 111 |
+
{"step": 198, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2913039409104114, "tokens": 120000, "cumulative_loss_tokens": 23760000, "grad_norm": 0.4921875, "lr": 3e-05, "finish_rate": 0.85, "comp_len": 472.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.9, "frames": {"chat": 254}, "mem_gb": 9.94}
|
| 112 |
+
{"step": 199, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4605057141344063, "tokens": 120000, "cumulative_loss_tokens": 23880000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.962, "comp_len": 381.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.6, "frames": {"chat": 315}, "mem_gb": 9.97}
|
| 113 |
+
{"step": 200, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.621974576252761, "tokens": 120000, "cumulative_loss_tokens": 24000000, "grad_norm": 0.62890625, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 300}, "mem_gb": 9.76}
|
| 114 |
+
[eval step 200] sample: 'To determine the rank of a matrix, we need to find the maximum number of linearly independent rows or columns. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly independent rows or col'
|
| 115 |
+
checkpoint snapshot queued -> outputs/healed/calib/glean_keep25_s1224/step0200
|
healed/calib/glean_keep25_s1224/wandb/run-20260717_210858-ypxqb0r8/logs/debug-core.log
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2026-07-17T21:08:58.389031686-07:00","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmppr5y76l8/port-399817.txt","pid":399817,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
|
| 2 |
+
{"time":"2026-07-17T21:08:58.389833134-07:00","level":"INFO","msg":"server: will exit if parent process dies","ppid":399817}
|
| 3 |
+
{"time":"2026-07-17T21:08:58.389779872-07:00","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-399817-399924-1676098820/socket","Net":"unix"}}
|
| 4 |
+
{"time":"2026-07-17T21:08:58.571304515-07:00","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
|
| 5 |
+
{"time":"2026-07-17T21:08:58.720383456-07:00","level":"INFO","msg":"handleInformInit: received","streamId":"ypxqb0r8","id":"1(@)"}
|
| 6 |
+
{"time":"2026-07-17T21:08:58.932318492-07:00","level":"INFO","msg":"handleInformInit: stream started","streamId":"ypxqb0r8","id":"1(@)"}
|
| 7 |
+
{"time":"2026-07-17T21:09:04.264002367-07:00","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"s0qspwiwts5x"}
|
| 8 |
+
{"time":"2026-07-17T22:31:39.159619829-07:00","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"s0qspwiwts5x"}
|
| 9 |
+
{"time":"2026-07-17T22:31:39.911767273-07:00","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"s0qspwiwts5x"}
|
| 10 |
+
{"time":"2026-07-17T22:31:39.915088596-07:00","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"ypxqb0r8","id":"1(@)"}
|
| 11 |
+
{"time":"2026-07-17T22:31:39.920683976-07:00","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"ypxqb0r8","id":"1(@)"}
|
| 12 |
+
{"time":"2026-07-17T22:31:40.552238826-07:00","level":"INFO","msg":"handleInformTeardown: server teardown initiated","id":"1(@)"}
|
| 13 |
+
{"time":"2026-07-17T22:31:40.552291018-07:00","level":"INFO","msg":"handleInformTeardown: server shutdown complete","id":"1(@)"}
|
| 14 |
+
{"time":"2026-07-17T22:31:40.552305578-07:00","level":"INFO","msg":"connection: closing","id":"1(@)"}
|
| 15 |
+
{"time":"2026-07-17T22:31:40.552345899-07:00","level":"INFO","msg":"server is shutting down"}
|
| 16 |
+
{"time":"2026-07-17T22:31:40.55236888-07:00","level":"INFO","msg":"connection: closed successfully","id":"1(@)"}
|
| 17 |
+
{"time":"2026-07-17T22:31:40.552388721-07:00","level":"INFO","msg":"processOutgoingData: finished","id":"1(@)"}
|
| 18 |
+
{"time":"2026-07-17T22:31:40.552401381-07:00","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"1(@)"}
|
| 19 |
+
{"time":"2026-07-17T22:31:40.552474414-07:00","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-399817-399924-1676098820/socket","Net":"unix"}}
|
| 20 |
+
{"time":"2026-07-17T22:31:40.552520895-07:00","level":"INFO","msg":"server is closed"}
|
healed/calib/glean_keep25_s1224/wandb/run-20260717_210858-ypxqb0r8/logs/debug.log
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17 21:08:58,718 INFO MainThread:399817 [wandb_setup.py:_flush():81] Current SDK version is 0.28.0
|
| 2 |
+
2026-07-17 21:08:58,718 INFO MainThread:399817 [wandb_setup.py:_flush():81] Configure stats pid to 399817
|
| 3 |
+
2026-07-17 21:08:58,718 INFO MainThread:399817 [wandb_setup.py:_flush():81] Loading settings from environment variables
|
| 4 |
+
2026-07-17 21:08:58,718 INFO MainThread:399817 [wandb_init.py:setup_run_log_directory():725] Logging user logs to outputs/healed/calib/glean_keep25_s1224/wandb/run-20260717_210858-ypxqb0r8/logs/debug.log
|
| 5 |
+
2026-07-17 21:08:58,718 INFO MainThread:399817 [wandb_init.py:setup_run_log_directory():726] Logging internal logs to outputs/healed/calib/glean_keep25_s1224/wandb/run-20260717_210858-ypxqb0r8/logs/debug-internal.log
|
| 6 |
+
2026-07-17 21:08:58,718 INFO MainThread:399817 [wandb_init.py:init():768] calling init triggers
|
| 7 |
+
2026-07-17 21:08:58,718 INFO MainThread:399817 [wandb_init.py:init():773] wandb.init called with sweep_config: {}
|
| 8 |
+
config: {'student': 'outputs/pruned/glean-0125inst-general-keep25', 'teacher': 'allenai/OLMoE-1B-7B-0125-Instruct', 'training_mode': 'off-policy', 'kl_direction': 'forward', 'dataset': 'allenai/RLVR-MATH', 'dataset_sources': None, 'max_difficulty': None, 'trajectories': 'outputs/teacher_trajectories/dolci_math_curated.jsonl', 'trajectory_dataset': 'allenai/Dolci-Instruct-RL', 'off_policy_frames': 'chat', 'off_policy_max_seq_len': 2048, 'topk_targets': 'outputs/teacher_trajectories/dolci_combined_top128', 'max_loss_tokens': None, 'loss_tokens_per_step': 120000, 'teacher_device': 'cuda:0', 'student_device': 'cuda:0', 'lr': 3e-05, 'optimizer': 'adamw8bit', 'weight_decay': 0.1, 'epochs': 9, 'prompts_per_step': 256, 'group_size': 1, 'rollout_batch': 64, 'micro_batch': 3, 'max_new_tokens': 256, 'max_prompt_len': 1024, 'warmup_steps': 10, 'max_grad_norm': 1.0, 'eval_every': 10, 'gsm8k_every': 0, 'gsm8k_n': 256, 'gsm8k_batch': 16, 'gsm8k_max_new_tokens': 512, 'gsm8k_frames': 'chat', 'save_every': 200, 'out_dir': 'outputs/healed/calib/glean_keep25_s1224', 'sweep': 200, 'wandb': True, 'wandb_project': 'glean-calib', 'wandb_run_name': 'calib_keep25_200', 'wandb_run_id': None, 'wandb_resume': None, 'wandb_mode': 'online', 'no_wandb_sync': False, 'debug': False, 'resume_from': 'outputs/healed/calib/glean_keep25_s1224/step0100', 'start_step': 0, 'no_grad_checkpointing': False, 'seed': 1224, 'no_teacher_overlap': False, 'sync_checkpoints': False, 'rollout_engine': 'hf', 'vllm_gpu': None, 'vllm_port': 8377, 'vllm_refresh_every': 5, 'vllm_serve_bin': 'vllm-plugin/.venv/bin/python', 'vllm_gpu_mem_util': 0.85, 'vllm_refresh_mode': 'reload', 'vllm_live_dir': None, 'resolved_kl_direction': 'forward', '_wandb': {}}
|
| 9 |
+
2026-07-17 21:08:58,718 INFO MainThread:399817 [wandb_init.py:init():816] starting backend
|
| 10 |
+
2026-07-17 21:08:58,718 INFO MainThread:399817 [wandb_init.py:init():831] sending inform_init request
|
| 11 |
+
2026-07-17 21:08:58,933 INFO MainThread:399817 [wandb_init.py:init():836] backend started and connected
|
| 12 |
+
2026-07-17 21:08:58,935 INFO MainThread:399817 [wandb_init.py:init():906] updated telemetry
|
| 13 |
+
2026-07-17 21:08:58,941 INFO MainThread:399817 [wandb_init.py:init():929] communicating run to backend with 90.0 second timeout
|
| 14 |
+
2026-07-17 21:08:59,120 INFO MainThread:399817 [wandb_init.py:init():974] starting run threads in backend
|
| 15 |
+
2026-07-17 21:08:59,254 INFO MainThread:399817 [wandb_run.py:_console_start():2523] atexit reg
|
| 16 |
+
2026-07-17 21:08:59,254 INFO MainThread:399817 [wandb_run.py:_redirect():2373] redirect: wrap_raw
|
| 17 |
+
2026-07-17 21:08:59,254 INFO MainThread:399817 [wandb_run.py:_redirect():2442] Wrapping output streams.
|
| 18 |
+
2026-07-17 21:08:59,254 INFO MainThread:399817 [wandb_run.py:_redirect():2465] Redirects installed.
|
| 19 |
+
2026-07-17 21:08:59,256 INFO MainThread:399817 [wandb_init.py:init():1012] run started, returning control to user process
|
| 20 |
+
2026-07-17 22:31:39,157 INFO MainThread:399817 [wandb_run.py:_finish():2285] finishing run hbfreed/glean-calib/ypxqb0r8
|
| 21 |
+
2026-07-17 22:31:39,158 INFO MainThread:399817 [wandb_run.py:_atexit_cleanup():2490] got exitcode: 0
|
| 22 |
+
2026-07-17 22:31:39,159 INFO MainThread:399817 [wandb_run.py:_restore():2472] restore
|
| 23 |
+
2026-07-17 22:31:39,159 INFO MainThread:399817 [wandb_run.py:_restore():2478] restore done
|
| 24 |
+
2026-07-17 22:31:39,914 INFO MainThread:399817 [wandb_run.py:_footer_sync_info():3895] logging synced files
|
healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/files/output.log
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
resumed student weights from outputs/healed/calib/glean_keep25_s1224/step0300 (fresh optimizer, step counter at 0)
|
| 2 |
+
58360 cached top-128 chat trajectories / 22,295,631 unique tokens | 185 steps/epoch | 400 total steps | student params 2.09B | teacher overlap=False
|
| 3 |
+
restored optimizer/scheduler state from step 300; rebuilt 228 paged buffers
|
| 4 |
+
{"step": 301, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.351279286532104, "tokens": 120000, "cumulative_loss_tokens": 36120000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.961, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 304}, "mem_gb": 9.71}
|
| 5 |
+
{"step": 302, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5407799015927439, "tokens": 120000, "cumulative_loss_tokens": 36240000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 321.7, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 51.3, "frames": {"chat": 373}, "mem_gb": 9.7}
|
| 6 |
+
{"step": 303, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3325376754036794, "tokens": 120000, "cumulative_loss_tokens": 36360000, "grad_norm": 0.53125, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 400.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.5, "frames": {"chat": 300}, "mem_gb": 9.98}
|
| 7 |
+
{"step": 304, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5965203912614534, "tokens": 120000, "cumulative_loss_tokens": 36480000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.982, "comp_len": 362.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 331}, "mem_gb": 9.82}
|
| 8 |
+
{"step": 305, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.27112077200754237, "tokens": 120000, "cumulative_loss_tokens": 36600000, "grad_norm": 0.458984375, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.4, "frames": {"chat": 259}, "mem_gb": 9.96}
|
| 9 |
+
{"step": 306, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.535515980368418, "tokens": 120000, "cumulative_loss_tokens": 36720000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 335.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.3, "frames": {"chat": 358}, "mem_gb": 9.58}
|
| 10 |
+
{"step": 307, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.6173073474912594, "tokens": 120000, "cumulative_loss_tokens": 36840000, "grad_norm": 0.66015625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 53.5, "frames": {"chat": 354}, "mem_gb": 9.84}
|
| 11 |
+
{"step": 308, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30061435020910576, "tokens": 120000, "cumulative_loss_tokens": 36960000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.939, "comp_len": 430.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.0, "frames": {"chat": 279}, "mem_gb": 9.9}
|
| 12 |
+
{"step": 309, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.622384964984407, "tokens": 120000, "cumulative_loss_tokens": 37080000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.9, "frames": {"chat": 354}, "mem_gb": 9.65}
|
| 13 |
+
{"step": 310, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.597136234145984, "tokens": 120000, "cumulative_loss_tokens": 37200000, "grad_norm": 0.6640625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 359.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 334}, "mem_gb": 9.86}
|
| 14 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 15 |
+
[eval step 310] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as follows:\n\n\\[\nA = \\begin{bmatrix}\n12 &"
|
| 16 |
+
{"step": 311, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5931549054012324, "tokens": 120000, "cumulative_loss_tokens": 37320000, "grad_norm": 0.63671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 367.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.2, "frames": {"chat": 327}, "mem_gb": 9.72}
|
| 17 |
+
{"step": 312, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5639030251913083, "tokens": 120000, "cumulative_loss_tokens": 37440000, "grad_norm": 0.6328125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 328.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.4, "frames": {"chat": 365}, "mem_gb": 9.82}
|
| 18 |
+
{"step": 313, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.31766464765102914, "tokens": 120000, "cumulative_loss_tokens": 37560000, "grad_norm": 0.5390625, "lr": 3e-05, "finish_rate": 0.869, "comp_len": 476.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.3, "frames": {"chat": 252}, "mem_gb": 10.03}
|
| 19 |
+
{"step": 314, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.25770912591886397, "tokens": 120000, "cumulative_loss_tokens": 37680000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 468.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.3, "frames": {"chat": 256}, "mem_gb": 9.99}
|
| 20 |
+
{"step": 315, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3189014506495868, "tokens": 120000, "cumulative_loss_tokens": 37800000, "grad_norm": 0.498046875, "lr": 3e-05, "finish_rate": 0.964, "comp_len": 397.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 43.1, "frames": {"chat": 302}, "mem_gb": 9.77}
|
| 21 |
+
{"step": 316, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5332931566566229, "tokens": 120000, "cumulative_loss_tokens": 37920000, "grad_norm": 0.65234375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 342.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.0, "frames": {"chat": 350}, "mem_gb": 9.67}
|
| 22 |
+
{"step": 317, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3288269461247449, "tokens": 120000, "cumulative_loss_tokens": 38040000, "grad_norm": 0.515625, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 411.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.9, "frames": {"chat": 292}, "mem_gb": 9.79}
|
| 23 |
+
{"step": 318, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5442317285993757, "tokens": 120000, "cumulative_loss_tokens": 38160000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 337.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.7, "frames": {"chat": 356}, "mem_gb": 9.6}
|
| 24 |
+
{"step": 319, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.383545723837552, "tokens": 120000, "cumulative_loss_tokens": 38280000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.977, "comp_len": 345.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 347}, "mem_gb": 9.77}
|
| 25 |
+
{"step": 320, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3346157630101157, "tokens": 120000, "cumulative_loss_tokens": 38400000, "grad_norm": 0.52734375, "lr": 3e-05, "finish_rate": 0.976, "comp_len": 363.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.6, "frames": {"chat": 330}, "mem_gb": 9.96}
|
| 26 |
+
[eval step 320] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as follows:\n\n\\[\nA = \\begin{bmatrix}\n12 &"
|
| 27 |
+
{"step": 321, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5644299145538981, "tokens": 120000, "cumulative_loss_tokens": 38520000, "grad_norm": 0.6171875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.5, "frames": {"chat": 363}, "mem_gb": 9.82}
|
| 28 |
+
{"step": 322, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5168023564506322, "tokens": 120000, "cumulative_loss_tokens": 38640000, "grad_norm": 0.625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 339.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 56.9, "frames": {"chat": 354}, "mem_gb": 9.68}
|
| 29 |
+
{"step": 323, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2756425909321134, "tokens": 120000, "cumulative_loss_tokens": 38760000, "grad_norm": 0.490234375, "lr": 3e-05, "finish_rate": 0.93, "comp_len": 444.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 45.6, "frames": {"chat": 270}, "mem_gb": 9.95}
|
| 30 |
+
{"step": 324, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5347250625066459, "tokens": 120000, "cumulative_loss_tokens": 38880000, "grad_norm": 0.60546875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 349.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.5, "frames": {"chat": 343}, "mem_gb": 9.72}
|
| 31 |
+
{"step": 325, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.13697400885016348, "tokens": 120000, "cumulative_loss_tokens": 39000000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 517.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.1, "frames": {"chat": 232}, "mem_gb": 9.91}
|
| 32 |
+
{"step": 326, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5018270760613183, "tokens": 120000, "cumulative_loss_tokens": 39120000, "grad_norm": 0.62109375, "lr": 3e-05, "finish_rate": 0.98, "comp_len": 346.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.2, "frames": {"chat": 346}, "mem_gb": 9.86}
|
| 33 |
+
{"step": 327, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.39994397104233503, "tokens": 120000, "cumulative_loss_tokens": 39240000, "grad_norm": 0.5625, "lr": 3e-05, "finish_rate": 0.951, "comp_len": 392.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.3, "frames": {"chat": 306}, "mem_gb": 9.95}
|
| 34 |
+
{"step": 328, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.26500454536170387, "tokens": 120000, "cumulative_loss_tokens": 39360000, "grad_norm": 0.44140625, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 439.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 40.3, "frames": {"chat": 273}, "mem_gb": 9.99}
|
| 35 |
+
{"step": 329, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.2892410087935937, "tokens": 120000, "cumulative_loss_tokens": 39480000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.959, "comp_len": 451.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 39.8, "frames": {"chat": 266}, "mem_gb": 9.82}
|
| 36 |
+
{"step": 330, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.4476620569754392, "tokens": 120000, "cumulative_loss_tokens": 39600000, "grad_norm": 0.59765625, "lr": 3e-05, "finish_rate": 0.936, "comp_len": 401.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.1, "frames": {"chat": 299}, "mem_gb": 10.01}
|
| 37 |
+
[eval step 330] sample: 'To compute the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nGiven the matrix:\n\\[\n\\begin{bmatrix}\n12 & -16 & 4 & 16 \\\\\n-9 & 11 & -1'
|
| 38 |
+
{"step": 331, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30612562722607206, "tokens": 120000, "cumulative_loss_tokens": 39720000, "grad_norm": 0.5078125, "lr": 3e-05, "finish_rate": 0.88, "comp_len": 463.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.2, "frames": {"chat": 259}, "mem_gb": 10.1}
|
| 39 |
+
{"step": 332, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3648861061472135, "tokens": 120000, "cumulative_loss_tokens": 39840000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.966, "comp_len": 369.2, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 46.3, "frames": {"chat": 325}, "mem_gb": 9.83}
|
| 40 |
+
{"step": 333, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.35644260462733607, "tokens": 120000, "cumulative_loss_tokens": 39960000, "grad_norm": 0.59765625, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 42.0, "frames": {"chat": 271}, "mem_gb": 9.91}
|
| 41 |
+
{"step": 334, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5318697121323397, "tokens": 120000, "cumulative_loss_tokens": 40080000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 334.3, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.8, "frames": {"chat": 359}, "mem_gb": 9.69}
|
| 42 |
+
{"step": 335, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5110903365676291, "tokens": 120000, "cumulative_loss_tokens": 40200000, "grad_norm": 0.59765625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 347.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.1, "frames": {"chat": 345}, "mem_gb": 9.75}
|
| 43 |
+
{"step": 336, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5708266700956971, "tokens": 120000, "cumulative_loss_tokens": 40320000, "grad_norm": 0.6640625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 301.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.4, "frames": {"chat": 398}, "mem_gb": 9.78}
|
| 44 |
+
{"step": 337, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3323912446099644, "tokens": 120000, "cumulative_loss_tokens": 40440000, "grad_norm": 0.5546875, "lr": 3e-05, "finish_rate": 0.964, "comp_len": 390.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.0, "frames": {"chat": 307}, "mem_gb": 9.98}
|
| 45 |
+
{"step": 338, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5195394132010949, "tokens": 120000, "cumulative_loss_tokens": 40560000, "grad_norm": 0.66015625, "lr": 3e-05, "finish_rate": 0.995, "comp_len": 308.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.0, "frames": {"chat": 389}, "mem_gb": 9.75}
|
| 46 |
+
{"step": 339, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.10721138076403489, "tokens": 120000, "cumulative_loss_tokens": 40680000, "grad_norm": 0.390625, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 519.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 37.6, "frames": {"chat": 231}, "mem_gb": 9.99}
|
| 47 |
+
{"step": 340, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5100581502144536, "tokens": 120000, "cumulative_loss_tokens": 40800000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.988, "comp_len": 373.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 48.5, "frames": {"chat": 321}, "mem_gb": 9.66}
|
| 48 |
+
[eval step 340] sample: 'To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns. A matrix is said to be of rank \\( r \\) if it has \\( r \\) linearly independent rows or col'
|
| 49 |
+
{"step": 341, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5770065285284072, "tokens": 120000, "cumulative_loss_tokens": 40920000, "grad_norm": 0.703125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 357.1, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 51.0, "frames": {"chat": 336}, "mem_gb": 9.69}
|
| 50 |
+
{"step": 342, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5449590804694531, "tokens": 120000, "cumulative_loss_tokens": 41040000, "grad_norm": 0.62890625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 355.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 50.0, "frames": {"chat": 338}, "mem_gb": 9.84}
|
| 51 |
+
{"step": 343, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.3079538588923402, "tokens": 120000, "cumulative_loss_tokens": 41160000, "grad_norm": 0.462890625, "lr": 3e-05, "finish_rate": 0.945, "comp_len": 436.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 44.4, "frames": {"chat": 275}, "mem_gb": 9.98}
|
| 52 |
+
{"step": 344, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.21514355221434187, "tokens": 120000, "cumulative_loss_tokens": 41280000, "grad_norm": 0.45703125, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 531.0, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 41.4, "frames": {"chat": 226}, "mem_gb": 10.01}
|
| 53 |
+
{"step": 345, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5597503078967954, "tokens": 120000, "cumulative_loss_tokens": 41400000, "grad_norm": 0.63671875, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 346.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 57.1, "frames": {"chat": 346}, "mem_gb": 9.8}
|
| 54 |
+
{"step": 346, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5308711614095917, "tokens": 120000, "cumulative_loss_tokens": 41520000, "grad_norm": 0.6015625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 330.6, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 363}, "mem_gb": 9.61}
|
| 55 |
+
{"step": 347, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.30469400957108783, "tokens": 120000, "cumulative_loss_tokens": 41640000, "grad_norm": 0.482421875, "lr": 3e-05, "finish_rate": 0.955, "comp_len": 383.4, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 47.4, "frames": {"chat": 313}, "mem_gb": 9.72}
|
| 56 |
+
{"step": 348, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5788642878273502, "tokens": 120000, "cumulative_loss_tokens": 41760000, "grad_norm": 0.61328125, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 323.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 52.6, "frames": {"chat": 371}, "mem_gb": 9.77}
|
| 57 |
+
{"step": 349, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.43604542372375726, "tokens": 120000, "cumulative_loss_tokens": 41880000, "grad_norm": 0.56640625, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 373.8, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 49.0, "frames": {"chat": 321}, "mem_gb": 9.55}
|
| 58 |
+
{"step": 350, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.5381324614154795, "tokens": 120000, "cumulative_loss_tokens": 42000000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.997, "comp_len": 365.9, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 54.2, "frames": {"chat": 328}, "mem_gb": 9.64}
|
| 59 |
+
[eval step 350] sample: "To find the rank of a matrix, we need to determine the maximum number of linearly independent rows or columns in the matrix. \n\nLet's represent the given matrix as follows:\n\n\\[\nA = \\begin{bmatrix}\n12 &"
|
healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/files/requirements.txt
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
certifi==2026.6.17
|
| 2 |
+
fonttools==4.63.0
|
| 3 |
+
urllib3==2.7.0
|
| 4 |
+
requests==2.34.2
|
| 5 |
+
nvidia-cublas-cu12==12.6.4.1
|
| 6 |
+
packaging==26.2
|
| 7 |
+
regex==2026.6.28
|
| 8 |
+
portalocker==3.2.0
|
| 9 |
+
safetensors==0.8.0
|
| 10 |
+
datasets==5.0.0
|
| 11 |
+
sqlitedict==2.1.0
|
| 12 |
+
narwhals==2.23.0
|
| 13 |
+
latex2sympy2_extended==1.11.0
|
| 14 |
+
smmap==5.0.3
|
| 15 |
+
colorama==0.4.6
|
| 16 |
+
anyio==4.14.1
|
| 17 |
+
cycler==0.12.1
|
| 18 |
+
pytablewriter==1.2.1
|
| 19 |
+
pydantic_core==2.46.4
|
| 20 |
+
huggingface_hub==0.36.2
|
| 21 |
+
charset-normalizer==3.4.8
|
| 22 |
+
nvidia-cusolver-cu12==11.7.1.2
|
| 23 |
+
joblib==1.5.3
|
| 24 |
+
typing-inspection==0.4.2
|
| 25 |
+
wandb==0.28.0
|
| 26 |
+
bitsandbytes==0.49.2
|
| 27 |
+
kiwisolver==1.5.0
|
| 28 |
+
cuda-toolkit==12.6.3
|
| 29 |
+
propcache==0.5.2
|
| 30 |
+
Jinja2==3.1.6
|
| 31 |
+
cuda-bindings==12.9.7
|
| 32 |
+
pluggy==1.6.0
|
| 33 |
+
pytest==9.1.1
|
| 34 |
+
fsspec==2026.4.0
|
| 35 |
+
tqdm==4.68.3
|
| 36 |
+
psutil==7.2.2
|
| 37 |
+
pydantic==2.13.4
|
| 38 |
+
typepy==1.3.5
|
| 39 |
+
xxhash==3.8.1
|
| 40 |
+
DataProperty==1.1.1
|
| 41 |
+
aiosignal==1.4.0
|
| 42 |
+
annotated-types==0.7.0
|
| 43 |
+
Pygments==2.20.0
|
| 44 |
+
zstandard==0.25.0
|
| 45 |
+
threadpoolctl==3.6.0
|
| 46 |
+
nvidia-cudnn-cu12==9.10.2.21
|
| 47 |
+
evaluate==0.4.6
|
| 48 |
+
pillow==12.3.0
|
| 49 |
+
GitPython==3.1.50
|
| 50 |
+
httpx==0.28.1
|
| 51 |
+
lm_eval==0.4.12
|
| 52 |
+
pytz==2026.2
|
| 53 |
+
mbstrdecoder==1.1.5
|
| 54 |
+
immutabledict==4.3.1
|
| 55 |
+
dill==0.4.1
|
| 56 |
+
mpmath==1.3.0
|
| 57 |
+
yarl==1.24.2
|
| 58 |
+
word2number==1.1
|
| 59 |
+
networkx==3.6.1
|
| 60 |
+
einops==0.8.2
|
| 61 |
+
tabulate==0.10.0
|
| 62 |
+
pathvalidate==3.3.1
|
| 63 |
+
tenacity==9.1.4
|
| 64 |
+
sentry-sdk==2.64.0
|
| 65 |
+
nvidia-nvshmem-cu12==3.4.5
|
| 66 |
+
hf-xet==1.5.1
|
| 67 |
+
six==1.17.0
|
| 68 |
+
pyarrow==24.0.0
|
| 69 |
+
nvidia-curand-cu12==10.3.7.77
|
| 70 |
+
cuda-pathfinder==1.5.6
|
| 71 |
+
sacrebleu==2.6.0
|
| 72 |
+
triton==3.7.1
|
| 73 |
+
frozenlist==1.8.0
|
| 74 |
+
antlr4-python3-runtime==4.11.0
|
| 75 |
+
tabledata==1.3.5
|
| 76 |
+
h11==0.16.0
|
| 77 |
+
nvidia-cusparselt-cu12==0.7.1
|
| 78 |
+
nvidia-nvjitlink-cu12==12.6.85
|
| 79 |
+
scipy==1.18.0
|
| 80 |
+
rouge_score==0.1.2
|
| 81 |
+
transformers==4.57.6
|
| 82 |
+
multiprocess==0.70.19
|
| 83 |
+
absl-py==2.5.0
|
| 84 |
+
accelerate==1.14.0
|
| 85 |
+
click==8.4.2
|
| 86 |
+
lxml==6.1.1
|
| 87 |
+
defusedxml==0.7.1
|
| 88 |
+
setuptools==81.0.0
|
| 89 |
+
nvidia-cuda-cupti-cu12==12.6.80
|
| 90 |
+
httpcore==1.0.9
|
| 91 |
+
langdetect==1.0.9
|
| 92 |
+
nltk==3.10.0
|
| 93 |
+
iniconfig==2.3.0
|
| 94 |
+
nvidia-cuda-runtime-cu12==12.6.77
|
| 95 |
+
nvidia-cusparse-cu12==12.5.4.2
|
| 96 |
+
matplotlib==3.11.0
|
| 97 |
+
protobuf==7.35.1
|
| 98 |
+
nvidia-cufft-cu12==11.3.0.4
|
| 99 |
+
pyparsing==3.3.2
|
| 100 |
+
chardet==6.0.0.post1
|
| 101 |
+
typing_extensions==4.16.0
|
| 102 |
+
stanford-stk==0.7.1
|
| 103 |
+
aiohappyeyeballs==2.7.1
|
| 104 |
+
MarkupSafe==3.0.3
|
| 105 |
+
nvidia-cufile-cu12==1.11.1.6
|
| 106 |
+
tcolorpy==0.1.7
|
| 107 |
+
scikit-learn==1.9.0
|
| 108 |
+
tokenizers==0.22.2
|
| 109 |
+
attrs==26.1.0
|
| 110 |
+
gitdb==4.0.12
|
| 111 |
+
aiohttp==3.14.1
|
| 112 |
+
glean==0.0.1
|
| 113 |
+
platformdirs==4.10.0
|
| 114 |
+
megablocks==0.11.0.dev0
|
| 115 |
+
nvidia-cuda-nvrtc-cu12==12.6.85
|
| 116 |
+
pandas==3.0.3
|
| 117 |
+
multidict==6.7.1
|
| 118 |
+
torch==2.12.1+cu126
|
| 119 |
+
contourpy==1.3.3
|
| 120 |
+
nvidia-nvtx-cu12==12.6.77
|
| 121 |
+
math-verify==0.9.0
|
| 122 |
+
filelock==3.29.5
|
| 123 |
+
idna==3.18
|
| 124 |
+
nvidia-nccl-cu12==2.29.3
|
| 125 |
+
python-dateutil==2.9.0.post0
|
| 126 |
+
more-itertools==11.1.0
|
| 127 |
+
numpy==2.0.2
|
| 128 |
+
sympy==1.14.0
|
| 129 |
+
PyYAML==6.0.3
|
healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/files/wandb-metadata.json
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"os": "Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35",
|
| 3 |
+
"python": "CPython 3.12.12",
|
| 4 |
+
"startedAt": "2026-07-18T06:57:39.078210Z",
|
| 5 |
+
"args": [
|
| 6 |
+
"--training-mode",
|
| 7 |
+
"off-policy",
|
| 8 |
+
"--kl-direction",
|
| 9 |
+
"forward",
|
| 10 |
+
"--topk-targets",
|
| 11 |
+
"outputs/teacher_trajectories/dolci_combined_top128",
|
| 12 |
+
"--off-policy-frames",
|
| 13 |
+
"chat",
|
| 14 |
+
"--student-device",
|
| 15 |
+
"cuda:0",
|
| 16 |
+
"--student",
|
| 17 |
+
"outputs/pruned/glean-0125inst-general-keep25",
|
| 18 |
+
"--resume-from",
|
| 19 |
+
"outputs/healed/calib/glean_keep25_s1224/step0300",
|
| 20 |
+
"--lr",
|
| 21 |
+
"3e-5",
|
| 22 |
+
"--optimizer",
|
| 23 |
+
"adamw8bit",
|
| 24 |
+
"--epochs",
|
| 25 |
+
"9",
|
| 26 |
+
"--sweep",
|
| 27 |
+
"400",
|
| 28 |
+
"--micro-batch",
|
| 29 |
+
"3",
|
| 30 |
+
"--loss-tokens-per-step",
|
| 31 |
+
"120000",
|
| 32 |
+
"--gsm8k-every",
|
| 33 |
+
"0",
|
| 34 |
+
"--save-every",
|
| 35 |
+
"400",
|
| 36 |
+
"--seed",
|
| 37 |
+
"1224",
|
| 38 |
+
"--out-dir",
|
| 39 |
+
"outputs/healed/calib/glean_keep25_s1224",
|
| 40 |
+
"--wandb",
|
| 41 |
+
"--wandb-mode",
|
| 42 |
+
"online",
|
| 43 |
+
"--wandb-project",
|
| 44 |
+
"glean-calib",
|
| 45 |
+
"--wandb-run-name",
|
| 46 |
+
"calib_keep25_400"
|
| 47 |
+
],
|
| 48 |
+
"program": "/home/henry/Documents/PythonProjects/variable-reap/scripts/11_distill_on_policy.py",
|
| 49 |
+
"codePath": "scripts/11_distill_on_policy.py",
|
| 50 |
+
"codePathLocal": "scripts/11_distill_on_policy.py",
|
| 51 |
+
"git": {
|
| 52 |
+
"remote": "https://github.com/hbfreed/variable-reap.git",
|
| 53 |
+
"commit": "247c7f0113d7a058b1db8044deddfd4456cc998d"
|
| 54 |
+
},
|
| 55 |
+
"email": "hbfreed@protonmail.com",
|
| 56 |
+
"root": "outputs/healed/calib/glean_keep25_s1224",
|
| 57 |
+
"host": "pop-os",
|
| 58 |
+
"executable": "/home/henry/Documents/PythonProjects/variable-reap/.venv/bin/python3",
|
| 59 |
+
"cpu_count": 24,
|
| 60 |
+
"cpu_count_logical": 48,
|
| 61 |
+
"gpu": "NVIDIA GeForce RTX 3090",
|
| 62 |
+
"gpu_count": 3,
|
| 63 |
+
"disk": {
|
| 64 |
+
"/": {
|
| 65 |
+
"total": "1958315118592",
|
| 66 |
+
"used": "1668439638016"
|
| 67 |
+
}
|
| 68 |
+
},
|
| 69 |
+
"memory": {
|
| 70 |
+
"total": "134900756480"
|
| 71 |
+
},
|
| 72 |
+
"gpu_nvidia": [
|
| 73 |
+
{
|
| 74 |
+
"name": "NVIDIA GeForce RTX 3090",
|
| 75 |
+
"memoryTotal": "25769803776",
|
| 76 |
+
"cudaCores": 10496,
|
| 77 |
+
"architecture": "Ampere",
|
| 78 |
+
"uuid": "GPU-8ca70870-ddf2-d274-bcc5-182c2075bced"
|
| 79 |
+
},
|
| 80 |
+
{
|
| 81 |
+
"name": "NVIDIA GeForce RTX 3090",
|
| 82 |
+
"memoryTotal": "25769803776",
|
| 83 |
+
"cudaCores": 10496,
|
| 84 |
+
"architecture": "Ampere",
|
| 85 |
+
"uuid": "GPU-a6acf07f-31f5-618f-a5d0-c0017e7e2e27"
|
| 86 |
+
},
|
| 87 |
+
{
|
| 88 |
+
"name": "NVIDIA GeForce RTX 3090",
|
| 89 |
+
"memoryTotal": "25769803776",
|
| 90 |
+
"cudaCores": 10496,
|
| 91 |
+
"architecture": "Ampere",
|
| 92 |
+
"uuid": "GPU-864c54df-0130-7780-e271-8a5551d1733f"
|
| 93 |
+
}
|
| 94 |
+
],
|
| 95 |
+
"cudaVersion": "13.0",
|
| 96 |
+
"writerId": "840j1ww7y2xmcobmkrrzrcafjpy0v2cg"
|
| 97 |
+
}
|
healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/logs/debug-core.log
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2026-07-17T23:57:38.746866474-07:00","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmp_vhp_drk/port-417245.txt","pid":417245,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
|
| 2 |
+
{"time":"2026-07-17T23:57:38.749038876-07:00","level":"INFO","msg":"server: will exit if parent process dies","ppid":417245}
|
| 3 |
+
{"time":"2026-07-17T23:57:38.748934082-07:00","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-417245-417348-471081702/socket","Net":"unix"}}
|
| 4 |
+
{"time":"2026-07-17T23:57:38.927226477-07:00","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
|
| 5 |
+
{"time":"2026-07-17T23:57:39.082122299-07:00","level":"INFO","msg":"handleInformInit: received","streamId":"5x8mg4as","id":"1(@)"}
|
| 6 |
+
{"time":"2026-07-17T23:57:39.29962067-07:00","level":"INFO","msg":"handleInformInit: stream started","streamId":"5x8mg4as","id":"1(@)"}
|
| 7 |
+
{"time":"2026-07-17T23:57:44.760027013-07:00","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"5pumthokbgux"}
|
healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/logs/debug-internal.log
ADDED
|
@@ -0,0 +1,335 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2026-07-17T23:57:39.082298694-07:00","level":"INFO","msg":"wandb-core"}
|
| 2 |
+
{"time":"2026-07-17T23:57:39.082803661-07:00","level":"INFO","msg":"stream: starting","core version":"0.28.0"}
|
| 3 |
+
{"time":"2026-07-17T23:57:39.299416513-07:00","level":"INFO","msg":"stream: created new stream","id":"5x8mg4as"}
|
| 4 |
+
{"time":"2026-07-17T23:57:39.299505736-07:00","level":"INFO","msg":"handler: started"}
|
| 5 |
+
{"time":"2026-07-17T23:57:39.29960975-07:00","level":"INFO","msg":"stream: started"}
|
| 6 |
+
{"time":"2026-07-17T23:57:39.299642731-07:00","level":"INFO","msg":"writer: started","stream_id":"5x8mg4as"}
|
| 7 |
+
{"time":"2026-07-17T23:57:39.299690392-07:00","level":"INFO","msg":"sender: started"}
|
| 8 |
+
{"time":"2026-07-17T23:57:43.114110592-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
|
| 9 |
+
{"time":"2026-07-17T23:57:43.36747684-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 10 |
+
{"time":"2026-07-17T23:57:58.115071634-07:00","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":0,"events_lines":2,"console_offset":1,"console_lines":2,"uploaded_len":2}
|
| 11 |
+
{"time":"2026-07-17T23:57:58.3042581-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 12 |
+
{"time":"2026-07-17T23:58:13.114489315-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":2,"events_lines":2}
|
| 13 |
+
{"time":"2026-07-17T23:58:13.354446239-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 14 |
+
{"time":"2026-07-17T23:58:28.114155095-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":4,"events_lines":2}
|
| 15 |
+
{"time":"2026-07-17T23:58:28.361670208-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 16 |
+
{"time":"2026-07-17T23:58:43.114668252-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":1,"events_offset":6,"events_lines":2,"console_offset":3,"console_lines":1}
|
| 17 |
+
{"time":"2026-07-17T23:58:43.380794731-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 18 |
+
{"time":"2026-07-17T23:58:58.114479326-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":8,"events_lines":2}
|
| 19 |
+
{"time":"2026-07-17T23:58:58.355010999-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 20 |
+
{"time":"2026-07-17T23:59:13.11428169-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":10,"events_lines":2}
|
| 21 |
+
{"time":"2026-07-17T23:59:13.36621911-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 22 |
+
{"time":"2026-07-17T23:59:28.114376993-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":12,"events_lines":2}
|
| 23 |
+
{"time":"2026-07-17T23:59:28.407411352-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 24 |
+
{"time":"2026-07-17T23:59:43.114722165-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":1,"history_lines":1,"events_offset":14,"events_lines":2,"console_offset":4,"console_lines":1}
|
| 25 |
+
{"time":"2026-07-17T23:59:43.36167679-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 26 |
+
{"time":"2026-07-17T23:59:58.11478371-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":16,"events_lines":2}
|
| 27 |
+
{"time":"2026-07-17T23:59:58.376816137-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 28 |
+
{"time":"2026-07-18T00:00:13.115050448-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":18,"events_lines":2}
|
| 29 |
+
{"time":"2026-07-18T00:00:13.337920114-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 30 |
+
{"time":"2026-07-18T00:00:28.115271535-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":2,"history_lines":1,"events_offset":20,"events_lines":2,"console_offset":5,"console_lines":1}
|
| 31 |
+
{"time":"2026-07-18T00:00:28.311223644-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 32 |
+
{"time":"2026-07-18T00:00:43.115012036-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":22,"events_lines":2}
|
| 33 |
+
{"time":"2026-07-18T00:00:43.328317707-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 34 |
+
{"time":"2026-07-18T00:00:58.114989365-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":24,"events_lines":2}
|
| 35 |
+
{"time":"2026-07-18T00:00:58.380397443-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 36 |
+
{"time":"2026-07-18T00:01:13.115005465-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":3,"history_lines":1,"events_offset":26,"events_lines":2,"console_offset":6,"console_lines":1}
|
| 37 |
+
{"time":"2026-07-18T00:01:13.372753281-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 38 |
+
{"time":"2026-07-18T00:01:28.114211478-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":28,"events_lines":2}
|
| 39 |
+
{"time":"2026-07-18T00:01:28.352751581-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 40 |
+
{"time":"2026-07-18T00:01:43.114903211-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":30,"events_lines":2}
|
| 41 |
+
{"time":"2026-07-18T00:01:43.346113242-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 42 |
+
{"time":"2026-07-18T00:01:58.114855479-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":4,"history_lines":1,"events_offset":32,"events_lines":2,"console_offset":7,"console_lines":1}
|
| 43 |
+
{"time":"2026-07-18T00:01:58.394214737-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 44 |
+
{"time":"2026-07-18T00:02:13.114507257-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":34,"events_lines":2}
|
| 45 |
+
{"time":"2026-07-18T00:02:13.35032796-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 46 |
+
{"time":"2026-07-18T00:02:28.114665431-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":36,"events_lines":2}
|
| 47 |
+
{"time":"2026-07-18T00:02:28.330604799-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 48 |
+
{"time":"2026-07-18T00:02:43.114218666-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":5,"history_lines":1,"events_offset":38,"events_lines":2,"console_offset":8,"console_lines":1}
|
| 49 |
+
{"time":"2026-07-18T00:02:43.349813792-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 50 |
+
{"time":"2026-07-18T00:02:58.114431912-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":40,"events_lines":2}
|
| 51 |
+
{"time":"2026-07-18T00:02:58.34796902-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 52 |
+
{"time":"2026-07-18T00:03:13.11497337-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":42,"events_lines":2}
|
| 53 |
+
{"time":"2026-07-18T00:03:13.371135223-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 54 |
+
{"time":"2026-07-18T00:03:28.114463482-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":44,"events_lines":2}
|
| 55 |
+
{"time":"2026-07-18T00:03:28.510351401-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 56 |
+
{"time":"2026-07-18T00:03:43.115158585-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":6,"history_lines":1,"events_offset":46,"events_lines":2,"console_offset":9,"console_lines":1}
|
| 57 |
+
{"time":"2026-07-18T00:03:43.356438258-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 58 |
+
{"time":"2026-07-18T00:03:58.114438721-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":48,"events_lines":2}
|
| 59 |
+
{"time":"2026-07-18T00:03:58.403746787-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 60 |
+
{"time":"2026-07-18T00:04:13.114996949-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":50,"events_lines":2}
|
| 61 |
+
{"time":"2026-07-18T00:04:13.319954444-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 62 |
+
{"time":"2026-07-18T00:04:28.114712109-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":7,"history_lines":1,"events_offset":52,"events_lines":2,"console_offset":10,"console_lines":1}
|
| 63 |
+
{"time":"2026-07-18T00:04:28.298069932-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 64 |
+
{"time":"2026-07-18T00:04:43.115088431-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":54,"events_lines":2}
|
| 65 |
+
{"time":"2026-07-18T00:04:43.343263752-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 66 |
+
{"time":"2026-07-18T00:04:58.114584924-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":56,"events_lines":2}
|
| 67 |
+
{"time":"2026-07-18T00:04:58.380467467-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 68 |
+
{"time":"2026-07-18T00:05:13.114212081-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":58,"events_lines":2}
|
| 69 |
+
{"time":"2026-07-18T00:05:13.321669939-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 70 |
+
{"time":"2026-07-18T00:05:28.114886002-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":8,"history_lines":1,"events_offset":60,"events_lines":2,"console_offset":11,"console_lines":1}
|
| 71 |
+
{"time":"2026-07-18T00:05:28.39485017-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 72 |
+
{"time":"2026-07-18T00:05:43.114921113-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":62,"events_lines":2}
|
| 73 |
+
{"time":"2026-07-18T00:05:43.291005047-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 74 |
+
{"time":"2026-07-18T00:05:58.114987824-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":64,"events_lines":2}
|
| 75 |
+
{"time":"2026-07-18T00:05:58.382197452-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 76 |
+
{"time":"2026-07-18T00:06:13.114951543-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":9,"history_lines":1,"events_offset":66,"events_lines":2,"console_offset":12,"console_lines":3}
|
| 77 |
+
{"time":"2026-07-18T00:06:13.338487941-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 78 |
+
{"time":"2026-07-18T00:06:28.114798568-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":68,"events_lines":2}
|
| 79 |
+
{"time":"2026-07-18T00:06:28.381525599-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 80 |
+
{"time":"2026-07-18T00:06:43.11428473-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":70,"events_lines":2}
|
| 81 |
+
{"time":"2026-07-18T00:06:43.38671635-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 82 |
+
{"time":"2026-07-18T00:06:58.114756205-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":10,"history_lines":1,"events_offset":72,"events_lines":2,"console_offset":15,"console_lines":1}
|
| 83 |
+
{"time":"2026-07-18T00:06:58.348834501-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 84 |
+
{"time":"2026-07-18T00:07:13.114292059-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":74,"events_lines":2}
|
| 85 |
+
{"time":"2026-07-18T00:07:13.425121915-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 86 |
+
{"time":"2026-07-18T00:07:28.114503866-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":76,"events_lines":2}
|
| 87 |
+
{"time":"2026-07-18T00:07:28.324174287-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 88 |
+
{"time":"2026-07-18T00:07:43.114974851-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":78,"events_lines":2}
|
| 89 |
+
{"time":"2026-07-18T00:07:43.347532967-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 90 |
+
{"time":"2026-07-18T00:07:58.114429613-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":11,"history_lines":1,"events_offset":80,"events_lines":2,"console_offset":16,"console_lines":1}
|
| 91 |
+
{"time":"2026-07-18T00:07:58.420549253-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 92 |
+
{"time":"2026-07-18T00:08:13.114227056-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":82,"events_lines":2}
|
| 93 |
+
{"time":"2026-07-18T00:08:13.33886902-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 94 |
+
{"time":"2026-07-18T00:08:28.115151465-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":84,"events_lines":2}
|
| 95 |
+
{"time":"2026-07-18T00:08:28.386022154-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 96 |
+
{"time":"2026-07-18T00:08:43.114901944-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":12,"history_lines":1,"events_offset":86,"events_lines":2,"console_offset":17,"console_lines":1}
|
| 97 |
+
{"time":"2026-07-18T00:08:43.452262134-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 98 |
+
{"time":"2026-07-18T00:08:58.114833119-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":88,"events_lines":2}
|
| 99 |
+
{"time":"2026-07-18T00:08:58.370403463-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 100 |
+
{"time":"2026-07-18T00:09:13.114378541-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":13,"history_lines":1,"events_offset":90,"events_lines":2,"console_offset":18,"console_lines":1}
|
| 101 |
+
{"time":"2026-07-18T00:09:13.349609065-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 102 |
+
{"time":"2026-07-18T00:09:28.114872174-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":92,"events_lines":2}
|
| 103 |
+
{"time":"2026-07-18T00:09:28.353673755-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 104 |
+
{"time":"2026-07-18T00:09:43.114776289-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":94,"events_lines":2}
|
| 105 |
+
{"time":"2026-07-18T00:09:43.354941715-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 106 |
+
{"time":"2026-07-18T00:09:58.114340831-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":14,"history_lines":1,"events_offset":96,"events_lines":2,"console_offset":19,"console_lines":1}
|
| 107 |
+
{"time":"2026-07-18T00:09:58.374027461-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 108 |
+
{"time":"2026-07-18T00:10:13.114556996-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":98,"events_lines":2}
|
| 109 |
+
{"time":"2026-07-18T00:10:13.375432485-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 110 |
+
{"time":"2026-07-18T00:10:28.115024578-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":100,"events_lines":2}
|
| 111 |
+
{"time":"2026-07-18T00:10:28.35050344-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 112 |
+
{"time":"2026-07-18T00:10:43.114732706-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":102,"events_lines":2}
|
| 113 |
+
{"time":"2026-07-18T00:10:43.299683562-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 114 |
+
{"time":"2026-07-18T00:10:58.115083454-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":15,"history_lines":1,"events_offset":104,"events_lines":2,"console_offset":20,"console_lines":1}
|
| 115 |
+
{"time":"2026-07-18T00:10:58.394849816-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 116 |
+
{"time":"2026-07-18T00:11:13.114197372-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":106,"events_lines":2}
|
| 117 |
+
{"time":"2026-07-18T00:11:13.356010573-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 118 |
+
{"time":"2026-07-18T00:11:28.114613943-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":108,"events_lines":2}
|
| 119 |
+
{"time":"2026-07-18T00:11:28.320183789-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 120 |
+
{"time":"2026-07-18T00:11:43.11460004-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":16,"history_lines":1,"events_offset":110,"events_lines":2,"console_offset":21,"console_lines":1}
|
| 121 |
+
{"time":"2026-07-18T00:11:43.347451675-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 122 |
+
{"time":"2026-07-18T00:11:58.114659189-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":112,"events_lines":2}
|
| 123 |
+
{"time":"2026-07-18T00:11:58.377632247-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 124 |
+
{"time":"2026-07-18T00:12:13.115019018-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":114,"events_lines":2}
|
| 125 |
+
{"time":"2026-07-18T00:12:13.308809625-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 126 |
+
{"time":"2026-07-18T00:12:28.114896081-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":116,"events_lines":2}
|
| 127 |
+
{"time":"2026-07-18T00:12:28.366914448-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 128 |
+
{"time":"2026-07-18T00:12:43.114832806-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":17,"history_lines":1,"events_offset":118,"events_lines":2,"console_offset":22,"console_lines":1}
|
| 129 |
+
{"time":"2026-07-18T00:12:43.40712624-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 130 |
+
{"time":"2026-07-18T00:12:58.114295675-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":120,"events_lines":2}
|
| 131 |
+
{"time":"2026-07-18T00:12:58.431232912-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 132 |
+
{"time":"2026-07-18T00:13:13.114781269-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":122,"events_lines":2}
|
| 133 |
+
{"time":"2026-07-18T00:13:13.311461101-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 134 |
+
{"time":"2026-07-18T00:13:28.115127707-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":18,"history_lines":1,"events_offset":124,"events_lines":2,"console_offset":23,"console_lines":1}
|
| 135 |
+
{"time":"2026-07-18T00:13:28.390737882-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 136 |
+
{"time":"2026-07-18T00:13:43.115094234-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":126,"events_lines":2}
|
| 137 |
+
{"time":"2026-07-18T00:13:43.380914006-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 138 |
+
{"time":"2026-07-18T00:13:58.115007558-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":128,"events_lines":2}
|
| 139 |
+
{"time":"2026-07-18T00:13:58.450051722-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 140 |
+
{"time":"2026-07-18T00:14:13.114721907-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":19,"history_lines":1,"events_offset":130,"events_lines":2,"console_offset":24,"console_lines":2}
|
| 141 |
+
{"time":"2026-07-18T00:14:13.343204958-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 142 |
+
{"time":"2026-07-18T00:14:28.114150635-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":132,"events_lines":2}
|
| 143 |
+
{"time":"2026-07-18T00:14:28.323429893-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 144 |
+
{"time":"2026-07-18T00:14:43.11441008-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":134,"events_lines":2}
|
| 145 |
+
{"time":"2026-07-18T00:14:43.41659135-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 146 |
+
{"time":"2026-07-18T00:14:58.114621394-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":136,"events_lines":2}
|
| 147 |
+
{"time":"2026-07-18T00:14:58.345904627-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 148 |
+
{"time":"2026-07-18T00:15:13.114536859-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":20,"history_lines":1,"events_offset":138,"events_lines":2,"console_offset":26,"console_lines":1}
|
| 149 |
+
{"time":"2026-07-18T00:15:13.415975374-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 150 |
+
{"time":"2026-07-18T00:15:28.114747723-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":140,"events_lines":2}
|
| 151 |
+
{"time":"2026-07-18T00:15:28.423126657-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 152 |
+
{"time":"2026-07-18T00:15:43.115072881-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":142,"events_lines":2}
|
| 153 |
+
{"time":"2026-07-18T00:15:43.358401111-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 154 |
+
{"time":"2026-07-18T00:15:58.11452616-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":144,"events_lines":2}
|
| 155 |
+
{"time":"2026-07-18T00:15:58.38758083-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 156 |
+
{"time":"2026-07-18T00:16:13.114448954-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":21,"history_lines":1,"events_offset":146,"events_lines":2,"console_offset":27,"console_lines":1}
|
| 157 |
+
{"time":"2026-07-18T00:16:13.366802202-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 158 |
+
{"time":"2026-07-18T00:16:28.114237865-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":148,"events_lines":2}
|
| 159 |
+
{"time":"2026-07-18T00:16:41.047068925-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 160 |
+
{"time":"2026-07-18T00:16:43.114429768-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":150,"events_lines":2}
|
| 161 |
+
{"time":"2026-07-18T00:16:50.768258887-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 162 |
+
{"time":"2026-07-18T00:16:58.114938902-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":22,"history_lines":1,"events_offset":152,"events_lines":2,"console_offset":28,"console_lines":1}
|
| 163 |
+
{"time":"2026-07-18T00:16:58.37032772-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 164 |
+
{"time":"2026-07-18T00:17:13.114459993-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":154,"events_lines":2}
|
| 165 |
+
{"time":"2026-07-18T00:17:13.353513603-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 166 |
+
{"time":"2026-07-18T00:17:28.115177124-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":156,"events_lines":2}
|
| 167 |
+
{"time":"2026-07-18T00:17:28.323644075-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 168 |
+
{"time":"2026-07-18T00:17:43.114474458-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":23,"history_lines":1,"events_offset":158,"events_lines":2,"console_offset":29,"console_lines":1}
|
| 169 |
+
{"time":"2026-07-18T00:17:43.40579956-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 170 |
+
{"time":"2026-07-18T00:17:58.114250988-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":160,"events_lines":2}
|
| 171 |
+
{"time":"2026-07-18T00:17:58.368057623-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 172 |
+
{"time":"2026-07-18T00:18:13.115061271-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":162,"events_lines":2}
|
| 173 |
+
{"time":"2026-07-18T00:18:13.378263557-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 174 |
+
{"time":"2026-07-18T00:18:28.114951854-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":24,"history_lines":1,"events_offset":164,"events_lines":2,"console_offset":30,"console_lines":1}
|
| 175 |
+
{"time":"2026-07-18T00:18:28.363443375-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 176 |
+
{"time":"2026-07-18T00:18:43.114418344-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":166,"events_lines":2}
|
| 177 |
+
{"time":"2026-07-18T00:18:43.32970248-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 178 |
+
{"time":"2026-07-18T00:18:58.114242345-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":168,"events_lines":2}
|
| 179 |
+
{"time":"2026-07-18T00:18:58.338862589-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 180 |
+
{"time":"2026-07-18T00:19:13.115157722-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":25,"history_lines":1,"events_offset":170,"events_lines":2,"console_offset":31,"console_lines":1}
|
| 181 |
+
{"time":"2026-07-18T00:19:13.357079566-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 182 |
+
{"time":"2026-07-18T00:19:28.114654793-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":172,"events_lines":2}
|
| 183 |
+
{"time":"2026-07-18T00:19:28.368258762-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 184 |
+
{"time":"2026-07-18T00:19:43.114684871-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":174,"events_lines":2}
|
| 185 |
+
{"time":"2026-07-18T00:19:43.433541451-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 186 |
+
{"time":"2026-07-18T00:19:58.114640846-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":26,"history_lines":1,"events_offset":176,"events_lines":2,"console_offset":32,"console_lines":1}
|
| 187 |
+
{"time":"2026-07-18T00:19:58.39659977-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 188 |
+
{"time":"2026-07-18T00:20:13.114380865-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":178,"events_lines":2}
|
| 189 |
+
{"time":"2026-07-18T00:20:13.33177363-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 190 |
+
{"time":"2026-07-18T00:20:28.114811786-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":180,"events_lines":2}
|
| 191 |
+
{"time":"2026-07-18T00:20:28.336992879-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 192 |
+
{"time":"2026-07-18T00:20:43.114868285-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":27,"history_lines":1,"events_offset":182,"events_lines":2,"console_offset":33,"console_lines":1}
|
| 193 |
+
{"time":"2026-07-18T00:20:43.360178121-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 194 |
+
{"time":"2026-07-18T00:20:58.114327704-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":184,"events_lines":2}
|
| 195 |
+
{"time":"2026-07-18T00:20:58.364339745-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 196 |
+
{"time":"2026-07-18T00:21:13.114487766-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":28,"history_lines":1,"events_offset":186,"events_lines":2,"console_offset":34,"console_lines":1}
|
| 197 |
+
{"time":"2026-07-18T00:21:13.453514791-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 198 |
+
{"time":"2026-07-18T00:21:28.114788873-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":188,"events_lines":2}
|
| 199 |
+
{"time":"2026-07-18T00:21:28.369748197-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 200 |
+
{"time":"2026-07-18T00:21:43.114283373-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":190,"events_lines":2}
|
| 201 |
+
{"time":"2026-07-18T00:21:43.357840871-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 202 |
+
{"time":"2026-07-18T00:21:58.114964322-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":29,"history_lines":1,"events_offset":192,"events_lines":2,"console_offset":35,"console_lines":2}
|
| 203 |
+
{"time":"2026-07-18T00:21:58.42313785-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 204 |
+
{"time":"2026-07-18T00:22:13.114602148-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":194,"events_lines":2}
|
| 205 |
+
{"time":"2026-07-18T00:22:13.382319232-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 206 |
+
{"time":"2026-07-18T00:22:28.115128742-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":196,"events_lines":2}
|
| 207 |
+
{"time":"2026-07-18T00:22:28.344518763-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 208 |
+
{"time":"2026-07-18T00:22:43.11485666-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":30,"history_lines":1,"events_offset":198,"events_lines":2,"console_offset":37,"console_lines":1}
|
| 209 |
+
{"time":"2026-07-18T00:22:43.352602136-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 210 |
+
{"time":"2026-07-18T00:22:58.115175868-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":200,"events_lines":2}
|
| 211 |
+
{"time":"2026-07-18T00:22:58.38586381-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 212 |
+
{"time":"2026-07-18T00:23:13.114748861-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":202,"events_lines":2}
|
| 213 |
+
{"time":"2026-07-18T00:23:13.322071404-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 214 |
+
{"time":"2026-07-18T00:23:28.114544511-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":31,"history_lines":1,"events_offset":204,"events_lines":2,"console_offset":38,"console_lines":1}
|
| 215 |
+
{"time":"2026-07-18T00:23:28.336230528-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 216 |
+
{"time":"2026-07-18T00:23:43.114445215-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":206,"events_lines":2}
|
| 217 |
+
{"time":"2026-07-18T00:23:43.343407252-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 218 |
+
{"time":"2026-07-18T00:23:58.115155935-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":208,"events_lines":2}
|
| 219 |
+
{"time":"2026-07-18T00:23:58.387638746-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 220 |
+
{"time":"2026-07-18T00:24:13.114964916-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":32,"history_lines":1,"events_offset":210,"events_lines":2,"console_offset":39,"console_lines":1}
|
| 221 |
+
{"time":"2026-07-18T00:24:13.372853896-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 222 |
+
{"time":"2026-07-18T00:24:28.1151713-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":212,"events_lines":2}
|
| 223 |
+
{"time":"2026-07-18T00:24:28.33694889-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 224 |
+
{"time":"2026-07-18T00:24:43.114523435-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":214,"events_lines":2}
|
| 225 |
+
{"time":"2026-07-18T00:24:43.426151617-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 226 |
+
{"time":"2026-07-18T00:24:58.114472151-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":33,"history_lines":1,"events_offset":216,"events_lines":2,"console_offset":40,"console_lines":1}
|
| 227 |
+
{"time":"2026-07-18T00:24:58.381332727-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 228 |
+
{"time":"2026-07-18T00:25:13.114936683-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":218,"events_lines":2}
|
| 229 |
+
{"time":"2026-07-18T00:25:13.344735038-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 230 |
+
{"time":"2026-07-18T00:25:28.114186105-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":220,"events_lines":2}
|
| 231 |
+
{"time":"2026-07-18T00:25:28.354739724-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 232 |
+
{"time":"2026-07-18T00:25:43.114241644-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":222,"events_lines":2}
|
| 233 |
+
{"time":"2026-07-18T00:25:43.347943497-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 234 |
+
{"time":"2026-07-18T00:25:58.114767273-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":34,"history_lines":1,"events_offset":224,"events_lines":2,"console_offset":41,"console_lines":1}
|
| 235 |
+
{"time":"2026-07-18T00:25:58.412158638-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 236 |
+
{"time":"2026-07-18T00:26:13.114420203-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":226,"events_lines":2}
|
| 237 |
+
{"time":"2026-07-18T00:26:13.401305454-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 238 |
+
{"time":"2026-07-18T00:26:28.114621231-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":228,"events_lines":2}
|
| 239 |
+
{"time":"2026-07-18T00:26:28.332453784-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 240 |
+
{"time":"2026-07-18T00:26:43.11515672-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":35,"history_lines":1,"events_offset":230,"events_lines":2,"console_offset":42,"console_lines":1}
|
| 241 |
+
{"time":"2026-07-18T00:26:43.366613387-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 242 |
+
{"time":"2026-07-18T00:26:58.11448949-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":232,"events_lines":2}
|
| 243 |
+
{"time":"2026-07-18T00:26:58.332852261-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 244 |
+
{"time":"2026-07-18T00:27:13.114501082-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":234,"events_lines":2}
|
| 245 |
+
{"time":"2026-07-18T00:27:13.332048486-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 246 |
+
{"time":"2026-07-18T00:27:28.114556786-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":36,"history_lines":1,"events_offset":236,"events_lines":2,"console_offset":43,"console_lines":1}
|
| 247 |
+
{"time":"2026-07-18T00:27:28.378220634-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 248 |
+
{"time":"2026-07-18T00:27:43.114499085-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":238,"events_lines":2}
|
| 249 |
+
{"time":"2026-07-18T00:27:43.389390272-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 250 |
+
{"time":"2026-07-18T00:27:58.114125675-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":240,"events_lines":2}
|
| 251 |
+
{"time":"2026-07-18T00:27:58.392616049-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 252 |
+
{"time":"2026-07-18T00:28:13.115005585-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":242,"events_lines":2}
|
| 253 |
+
{"time":"2026-07-18T00:28:13.364769596-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 254 |
+
{"time":"2026-07-18T00:28:28.114784719-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":37,"history_lines":1,"events_offset":244,"events_lines":2,"console_offset":44,"console_lines":1}
|
| 255 |
+
{"time":"2026-07-18T00:28:28.357918003-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 256 |
+
{"time":"2026-07-18T00:28:43.114857503-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":246,"events_lines":2}
|
| 257 |
+
{"time":"2026-07-18T00:28:43.322113679-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 258 |
+
{"time":"2026-07-18T00:28:58.114276606-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":38,"history_lines":1,"events_offset":248,"events_lines":2,"console_offset":45,"console_lines":1}
|
| 259 |
+
{"time":"2026-07-18T00:28:58.444313743-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 260 |
+
{"time":"2026-07-18T00:29:13.114955549-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":250,"events_lines":2}
|
| 261 |
+
{"time":"2026-07-18T00:29:13.35555949-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 262 |
+
{"time":"2026-07-18T00:29:28.114553048-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":252,"events_lines":2}
|
| 263 |
+
{"time":"2026-07-18T00:29:28.413704061-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 264 |
+
{"time":"2026-07-18T00:29:43.114322482-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":254,"events_lines":2}
|
| 265 |
+
{"time":"2026-07-18T00:29:43.369003235-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 266 |
+
{"time":"2026-07-18T00:29:58.114355874-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":39,"history_lines":1,"events_offset":256,"events_lines":2,"console_offset":46,"console_lines":2}
|
| 267 |
+
{"time":"2026-07-18T00:29:58.279112905-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 268 |
+
{"time":"2026-07-18T00:30:13.114539572-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":258,"events_lines":2}
|
| 269 |
+
{"time":"2026-07-18T00:30:13.403241312-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 270 |
+
{"time":"2026-07-18T00:30:28.114324117-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":260,"events_lines":2}
|
| 271 |
+
{"time":"2026-07-18T00:30:28.36840901-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 272 |
+
{"time":"2026-07-18T00:30:43.114781193-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":40,"history_lines":1,"events_offset":262,"events_lines":2,"console_offset":48,"console_lines":1}
|
| 273 |
+
{"time":"2026-07-18T00:30:43.398600393-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 274 |
+
{"time":"2026-07-18T00:30:58.114661161-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":264,"events_lines":2}
|
| 275 |
+
{"time":"2026-07-18T00:30:58.345818561-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 276 |
+
{"time":"2026-07-18T00:31:13.114156006-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":266,"events_lines":2}
|
| 277 |
+
{"time":"2026-07-18T00:31:13.378968761-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 278 |
+
{"time":"2026-07-18T00:31:28.114444997-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":268,"events_lines":2}
|
| 279 |
+
{"time":"2026-07-18T00:31:28.436054957-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 280 |
+
{"time":"2026-07-18T00:31:43.114549521-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":1,"events_offset":270,"events_lines":2,"console_offset":49,"console_lines":1}
|
| 281 |
+
{"time":"2026-07-18T00:31:43.352325299-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 282 |
+
{"time":"2026-07-18T00:31:58.114212262-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":272,"events_lines":2}
|
| 283 |
+
{"time":"2026-07-18T00:31:58.362499915-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 284 |
+
{"time":"2026-07-18T00:32:13.114112251-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":274,"events_lines":2}
|
| 285 |
+
{"time":"2026-07-18T00:32:13.356680496-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 286 |
+
{"time":"2026-07-18T00:32:28.114704512-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":42,"history_lines":1,"events_offset":276,"events_lines":2,"console_offset":50,"console_lines":1}
|
| 287 |
+
{"time":"2026-07-18T00:32:28.414907739-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 288 |
+
{"time":"2026-07-18T00:32:43.115086876-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":278,"events_lines":2}
|
| 289 |
+
{"time":"2026-07-18T00:32:43.32505073-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 290 |
+
{"time":"2026-07-18T00:32:58.11457102-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":43,"history_lines":1,"events_offset":280,"events_lines":2,"console_offset":51,"console_lines":1}
|
| 291 |
+
{"time":"2026-07-18T00:32:58.422558764-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 292 |
+
{"time":"2026-07-18T00:33:13.114309703-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":282,"events_lines":2}
|
| 293 |
+
{"time":"2026-07-18T00:33:13.319519991-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 294 |
+
{"time":"2026-07-18T00:33:28.115009927-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":284,"events_lines":2}
|
| 295 |
+
{"time":"2026-07-18T00:33:28.328599711-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 296 |
+
{"time":"2026-07-18T00:33:43.114340757-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":286,"events_lines":2}
|
| 297 |
+
{"time":"2026-07-18T00:33:43.376808316-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 298 |
+
{"time":"2026-07-18T00:33:58.114748932-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":44,"history_lines":1,"events_offset":288,"events_lines":2,"console_offset":52,"console_lines":1}
|
| 299 |
+
{"time":"2026-07-18T00:33:58.37598167-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 300 |
+
{"time":"2026-07-18T00:34:13.115103685-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":290,"events_lines":2}
|
| 301 |
+
{"time":"2026-07-18T00:34:13.373147588-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 302 |
+
{"time":"2026-07-18T00:34:28.114633361-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":292,"events_lines":2}
|
| 303 |
+
{"time":"2026-07-18T00:34:28.410383332-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 304 |
+
{"time":"2026-07-18T00:34:43.114404145-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":294,"events_lines":2}
|
| 305 |
+
{"time":"2026-07-18T00:34:43.360548757-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 306 |
+
{"time":"2026-07-18T00:34:58.11481105-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":45,"history_lines":1,"events_offset":296,"events_lines":2,"console_offset":53,"console_lines":1}
|
| 307 |
+
{"time":"2026-07-18T00:34:58.368686176-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 308 |
+
{"time":"2026-07-18T00:35:13.114615074-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":298,"events_lines":2}
|
| 309 |
+
{"time":"2026-07-18T00:35:13.424068036-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 310 |
+
{"time":"2026-07-18T00:35:28.11503666-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":300,"events_lines":2}
|
| 311 |
+
{"time":"2026-07-18T00:35:28.355005379-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 312 |
+
{"time":"2026-07-18T00:35:43.114608837-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":46,"history_lines":1,"events_offset":302,"events_lines":2,"console_offset":54,"console_lines":1}
|
| 313 |
+
{"time":"2026-07-18T00:35:43.412252981-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 314 |
+
{"time":"2026-07-18T00:35:58.114767114-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":304,"events_lines":2}
|
| 315 |
+
{"time":"2026-07-18T00:35:58.348510919-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 316 |
+
{"time":"2026-07-18T00:36:13.114162146-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":306,"events_lines":2}
|
| 317 |
+
{"time":"2026-07-18T00:36:13.359668117-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 318 |
+
{"time":"2026-07-18T00:36:28.11485477-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":47,"history_lines":1,"events_offset":308,"events_lines":2,"console_offset":55,"console_lines":1}
|
| 319 |
+
{"time":"2026-07-18T00:36:28.41786534-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 320 |
+
{"time":"2026-07-18T00:36:43.114117448-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":310,"events_lines":2}
|
| 321 |
+
{"time":"2026-07-18T00:36:43.40898333-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 322 |
+
{"time":"2026-07-18T00:36:58.114940316-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":312,"events_lines":2}
|
| 323 |
+
{"time":"2026-07-18T00:36:58.414177222-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 324 |
+
{"time":"2026-07-18T00:37:13.11470599-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":314,"events_lines":2}
|
| 325 |
+
{"time":"2026-07-18T00:37:13.404410923-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 326 |
+
{"time":"2026-07-18T00:37:28.115211578-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":48,"history_lines":1,"events_offset":316,"events_lines":2,"console_offset":56,"console_lines":1}
|
| 327 |
+
{"time":"2026-07-18T00:37:28.463553977-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 328 |
+
{"time":"2026-07-18T00:37:43.114237328-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":318,"events_lines":2}
|
| 329 |
+
{"time":"2026-07-18T00:37:43.38473356-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 330 |
+
{"time":"2026-07-18T00:37:58.115021065-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":320,"events_lines":2}
|
| 331 |
+
{"time":"2026-07-18T00:37:58.418948895-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 332 |
+
{"time":"2026-07-18T00:38:13.114946814-07:00","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":49,"history_lines":1,"events_offset":322,"events_lines":2,"console_offset":57,"console_lines":2}
|
| 333 |
+
{"time":"2026-07-18T00:38:13.421072086-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 334 |
+
{"time":"2026-07-18T00:38:28.114884044-07:00","level":"INFO","msg":"filestream: sending request","total_files":1,"events_offset":324,"events_lines":2}
|
| 335 |
+
{"time":"2026-07-18T00:38:28.320243707-07:00","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/logs/debug.log
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17 23:57:39,079 INFO MainThread:417245 [wandb_setup.py:_flush():81] Current SDK version is 0.28.0
|
| 2 |
+
2026-07-17 23:57:39,080 INFO MainThread:417245 [wandb_setup.py:_flush():81] Configure stats pid to 417245
|
| 3 |
+
2026-07-17 23:57:39,080 INFO MainThread:417245 [wandb_setup.py:_flush():81] Loading settings from environment variables
|
| 4 |
+
2026-07-17 23:57:39,080 INFO MainThread:417245 [wandb_init.py:setup_run_log_directory():725] Logging user logs to outputs/healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/logs/debug.log
|
| 5 |
+
2026-07-17 23:57:39,080 INFO MainThread:417245 [wandb_init.py:setup_run_log_directory():726] Logging internal logs to outputs/healed/calib/glean_keep25_s1224/wandb/run-20260717_235739-5x8mg4as/logs/debug-internal.log
|
| 6 |
+
2026-07-17 23:57:39,080 INFO MainThread:417245 [wandb_init.py:init():768] calling init triggers
|
| 7 |
+
2026-07-17 23:57:39,080 INFO MainThread:417245 [wandb_init.py:init():773] wandb.init called with sweep_config: {}
|
| 8 |
+
config: {'student': 'outputs/pruned/glean-0125inst-general-keep25', 'teacher': 'allenai/OLMoE-1B-7B-0125-Instruct', 'training_mode': 'off-policy', 'kl_direction': 'forward', 'dataset': 'allenai/RLVR-MATH', 'dataset_sources': None, 'max_difficulty': None, 'trajectories': 'outputs/teacher_trajectories/dolci_math_curated.jsonl', 'trajectory_dataset': 'allenai/Dolci-Instruct-RL', 'off_policy_frames': 'chat', 'off_policy_max_seq_len': 2048, 'topk_targets': 'outputs/teacher_trajectories/dolci_combined_top128', 'max_loss_tokens': None, 'loss_tokens_per_step': 120000, 'teacher_device': 'cuda:0', 'student_device': 'cuda:0', 'lr': 3e-05, 'optimizer': 'adamw8bit', 'weight_decay': 0.1, 'epochs': 9, 'prompts_per_step': 256, 'group_size': 1, 'rollout_batch': 64, 'micro_batch': 3, 'max_new_tokens': 256, 'max_prompt_len': 1024, 'warmup_steps': 10, 'max_grad_norm': 1.0, 'eval_every': 10, 'gsm8k_every': 0, 'gsm8k_n': 256, 'gsm8k_batch': 16, 'gsm8k_max_new_tokens': 512, 'gsm8k_frames': 'chat', 'save_every': 400, 'out_dir': 'outputs/healed/calib/glean_keep25_s1224', 'sweep': 400, 'wandb': True, 'wandb_project': 'glean-calib', 'wandb_run_name': 'calib_keep25_400', 'wandb_run_id': None, 'wandb_resume': None, 'wandb_mode': 'online', 'no_wandb_sync': False, 'debug': False, 'resume_from': 'outputs/healed/calib/glean_keep25_s1224/step0300', 'start_step': 0, 'no_grad_checkpointing': False, 'seed': 1224, 'no_teacher_overlap': False, 'sync_checkpoints': False, 'rollout_engine': 'hf', 'vllm_gpu': None, 'vllm_port': 8377, 'vllm_refresh_every': 5, 'vllm_serve_bin': 'vllm-plugin/.venv/bin/python', 'vllm_gpu_mem_util': 0.85, 'vllm_refresh_mode': 'reload', 'vllm_live_dir': None, 'resolved_kl_direction': 'forward', '_wandb': {}}
|
| 9 |
+
2026-07-17 23:57:39,080 INFO MainThread:417245 [wandb_init.py:init():816] starting backend
|
| 10 |
+
2026-07-17 23:57:39,080 INFO MainThread:417245 [wandb_init.py:init():831] sending inform_init request
|
| 11 |
+
2026-07-17 23:57:39,300 INFO MainThread:417245 [wandb_init.py:init():836] backend started and connected
|
| 12 |
+
2026-07-17 23:57:39,302 INFO MainThread:417245 [wandb_init.py:init():906] updated telemetry
|
| 13 |
+
2026-07-17 23:57:39,308 INFO MainThread:417245 [wandb_init.py:init():929] communicating run to backend with 90.0 second timeout
|
| 14 |
+
2026-07-17 23:57:39,616 INFO MainThread:417245 [wandb_init.py:init():974] starting run threads in backend
|
| 15 |
+
2026-07-17 23:57:39,749 INFO MainThread:417245 [wandb_run.py:_console_start():2523] atexit reg
|
| 16 |
+
2026-07-17 23:57:39,749 INFO MainThread:417245 [wandb_run.py:_redirect():2373] redirect: wrap_raw
|
| 17 |
+
2026-07-17 23:57:39,749 INFO MainThread:417245 [wandb_run.py:_redirect():2442] Wrapping output streams.
|
| 18 |
+
2026-07-17 23:57:39,749 INFO MainThread:417245 [wandb_run.py:_redirect():2465] Redirects installed.
|
| 19 |
+
2026-07-17 23:57:39,751 INFO MainThread:417245 [wandb_init.py:init():1012] run started, returning control to user process
|
healed/calib/glean_keep50_s1224.console.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/calib/glean_keep50_s1224.eval.log
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 1 |
37%|███▋ | 187/500 [00:00<00:00, 1863.77it/s]
|
| 2 |
77%|███████▋ | 383/500 [00:00<00:00, 1918.66it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 4 |
37%|███▋ | 186/500 [00:00<00:00, 1855.94it/s]
|
| 5 |
76%|███████▌ | 380/500 [00:00<00:00, 1903.22it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 6 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 7 |
38%|███▊ | 188/500 [00:00<00:00, 1873.78it/s]
|
| 8 |
77%|███████▋ | 383/500 [00:00<00:00, 1916.13it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T21:12:38-07:00 serving outputs/healed/calib/glean_keep50_s1224/step0100 on GPU 1 port 8431
|
| 2 |
+
2026-07-17T21:12:38-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T21:13:08-07:00 server up; chat pass [gsm8k_cot_zeroshot]
|
| 4 |
+
2026-07-17:21:13:09 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 5 |
+
2026-07-17:21:13:16 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot']
|
| 6 |
+
2026-07-17:21:13:17 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 7 |
+
2026-07-17:21:13:17 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding!
|
| 8 |
+
2026-07-17:21:13:17 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8431/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 9 |
+
2026-07-17:21:13:17 INFO [models.api_models:179] Using max length 2048 - 1
|
| 10 |
+
2026-07-17:21:13:17 INFO [models.api_models:200] Using tokenizer None
|
| 11 |
+
2026-07-17:21:13:19 INFO [evaluator_utils:446] Selected tasks:
|
| 12 |
+
2026-07-17:21:13:19 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 13 |
+
2026-07-17:21:13:19 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280}
|
| 14 |
+
2026-07-17:21:13:19 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 15 |
+
|
| 16 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 17 |
37%|███▋ | 187/500 [00:00<00:00, 1863.77it/s]
|
| 18 |
77%|███████▋ | 383/500 [00:00<00:00, 1918.66it/s]
|
| 19 |
+
2026-07-17:21:13:19 INFO [evaluator:585] Running generate_until requests
|
| 20 |
+
2026-07-17:21:13:19 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 21 |
+
|
| 22 |
+
2026-07-17:21:14:20 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 23 |
+
2026-07-17:21:14:20 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/calib/keep50_step100/student/*.jsonl
|
| 24 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8431/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({'max_gen_toks': 1280}), limit: 500.0, num_fewshot: None, batch_size: 1
|
| 25 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 26 |
+
|------------------|------:|----------------|-----:|-----------|---|----:|---|-----:|
|
| 27 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match|↑ |0.558|± |0.0222|
|
| 28 |
+
| | |strict-match | 0|exact_match|↑ |0.000|± |0.0000|
|
| 29 |
+
|
| 30 |
+
2026-07-17T21:14:21-07:00 lm_eval exit=0 -> outputs/evals/general_suite/calib/keep50_step100
|
| 31 |
+
2026-07-17T22:41:41-07:00 serving outputs/healed/calib/glean_keep50_s1224/step0200 on GPU 1 port 8431
|
| 32 |
+
2026-07-17T22:41:41-07:00 waiting for server /health ...
|
| 33 |
+
2026-07-17T22:42:11-07:00 server up; chat pass [gsm8k_cot_zeroshot]
|
| 34 |
+
2026-07-17:22:42:11 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 35 |
+
2026-07-17:22:42:18 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot']
|
| 36 |
+
2026-07-17:22:42:19 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 37 |
+
2026-07-17:22:42:19 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding!
|
| 38 |
+
2026-07-17:22:42:19 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8431/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 39 |
+
2026-07-17:22:42:19 INFO [models.api_models:179] Using max length 2048 - 1
|
| 40 |
+
2026-07-17:22:42:19 INFO [models.api_models:200] Using tokenizer None
|
| 41 |
+
2026-07-17:22:42:21 INFO [evaluator_utils:446] Selected tasks:
|
| 42 |
+
2026-07-17:22:42:21 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 43 |
+
2026-07-17:22:42:21 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280}
|
| 44 |
+
2026-07-17:22:42:21 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 45 |
+
|
| 46 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 47 |
37%|███▋ | 186/500 [00:00<00:00, 1855.94it/s]
|
| 48 |
76%|███████▌ | 380/500 [00:00<00:00, 1903.22it/s]
|
| 49 |
+
2026-07-17:22:42:21 INFO [evaluator:585] Running generate_until requests
|
| 50 |
+
2026-07-17:22:42:21 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 51 |
+
|
| 52 |
+
2026-07-17:22:43:17 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 53 |
+
2026-07-17:22:43:17 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/calib/keep50_step200/student/*.jsonl
|
| 54 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8431/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({'max_gen_toks': 1280}), limit: 500.0, num_fewshot: None, batch_size: 1
|
| 55 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 56 |
+
|------------------|------:|----------------|-----:|-----------|---|----:|---|-----:|
|
| 57 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match|↑ |0.564|± |0.0222|
|
| 58 |
+
| | |strict-match | 0|exact_match|↑ |0.000|± |0.0000|
|
| 59 |
+
|
| 60 |
+
2026-07-17T22:43:18-07:00 lm_eval exit=0 -> outputs/evals/general_suite/calib/keep50_step200
|
| 61 |
+
2026-07-18T00:09:57-07:00 serving outputs/healed/calib/glean_keep50_s1224/step0300 on GPU 1 port 8431
|
| 62 |
+
2026-07-18T00:09:57-07:00 waiting for server /health ...
|
| 63 |
+
2026-07-18T00:10:27-07:00 server up; chat pass [gsm8k_cot_zeroshot]
|
| 64 |
+
2026-07-18:00:10:28 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 65 |
+
2026-07-18:00:10:34 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot']
|
| 66 |
+
2026-07-18:00:10:36 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 67 |
+
2026-07-18:00:10:36 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding!
|
| 68 |
+
2026-07-18:00:10:36 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8431/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 69 |
+
2026-07-18:00:10:36 INFO [models.api_models:179] Using max length 2048 - 1
|
| 70 |
+
2026-07-18:00:10:36 INFO [models.api_models:200] Using tokenizer None
|
| 71 |
+
2026-07-18:00:10:37 INFO [evaluator_utils:446] Selected tasks:
|
| 72 |
+
2026-07-18:00:10:37 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 73 |
+
2026-07-18:00:10:37 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280}
|
| 74 |
+
2026-07-18:00:10:37 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 75 |
+
|
| 76 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 77 |
38%|███▊ | 188/500 [00:00<00:00, 1873.78it/s]
|
| 78 |
77%|███████▋ | 383/500 [00:00<00:00, 1916.13it/s]
|
| 79 |
+
2026-07-18:00:10:38 INFO [evaluator:585] Running generate_until requests
|
| 80 |
+
2026-07-18:00:10:38 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 81 |
+
|
| 82 |
+
2026-07-18:00:11:36 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 83 |
+
2026-07-18:00:11:36 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/calib/keep50_step300/student/*.jsonl
|
| 84 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8431/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({'max_gen_toks': 1280}), limit: 500.0, num_fewshot: None, batch_size: 1
|
| 85 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 86 |
+
|------------------|------:|----------------|-----:|-----------|---|----:|---|-----:|
|
| 87 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match|↑ |0.566|± |0.0222|
|
| 88 |
+
| | |strict-match | 0|exact_match|↑ |0.000|± |0.0000|
|
| 89 |
+
|
| 90 |
+
2026-07-18T00:11:37-07:00 lm_eval exit=0 -> outputs/evals/general_suite/calib/keep50_step300
|
healed/calib/glean_keep50_s1224/args.json
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"student": "outputs/pruned/glean-0125inst-general-keep50",
|
| 3 |
+
"teacher": "allenai/OLMoE-1B-7B-0125-Instruct",
|
| 4 |
+
"training_mode": "off-policy",
|
| 5 |
+
"kl_direction": "forward",
|
| 6 |
+
"dataset": "allenai/RLVR-MATH",
|
| 7 |
+
"dataset_sources": null,
|
| 8 |
+
"max_difficulty": null,
|
| 9 |
+
"trajectories": "outputs/teacher_trajectories/dolci_math_curated.jsonl",
|
| 10 |
+
"trajectory_dataset": "allenai/Dolci-Instruct-RL",
|
| 11 |
+
"off_policy_frames": "chat",
|
| 12 |
+
"off_policy_max_seq_len": 2048,
|
| 13 |
+
"topk_targets": "outputs/teacher_trajectories/dolci_combined_top128",
|
| 14 |
+
"max_loss_tokens": null,
|
| 15 |
+
"loss_tokens_per_step": 120000,
|
| 16 |
+
"teacher_device": "cuda:0",
|
| 17 |
+
"student_device": "cuda:0",
|
| 18 |
+
"lr": 3e-05,
|
| 19 |
+
"optimizer": "adamw8bit",
|
| 20 |
+
"weight_decay": 0.1,
|
| 21 |
+
"epochs": 9,
|
| 22 |
+
"prompts_per_step": 256,
|
| 23 |
+
"group_size": 1,
|
| 24 |
+
"rollout_batch": 64,
|
| 25 |
+
"micro_batch": 3,
|
| 26 |
+
"max_new_tokens": 256,
|
| 27 |
+
"max_prompt_len": 1024,
|
| 28 |
+
"warmup_steps": 10,
|
| 29 |
+
"max_grad_norm": 1.0,
|
| 30 |
+
"eval_every": 10,
|
| 31 |
+
"gsm8k_every": 0,
|
| 32 |
+
"gsm8k_n": 256,
|
| 33 |
+
"gsm8k_batch": 16,
|
| 34 |
+
"gsm8k_max_new_tokens": 512,
|
| 35 |
+
"gsm8k_frames": "chat",
|
| 36 |
+
"save_every": 400,
|
| 37 |
+
"out_dir": "outputs/healed/calib/glean_keep50_s1224",
|
| 38 |
+
"sweep": 400,
|
| 39 |
+
"wandb": true,
|
| 40 |
+
"wandb_project": "glean-calib",
|
| 41 |
+
"wandb_run_name": "calib_keep50_400",
|
| 42 |
+
"wandb_run_id": null,
|
| 43 |
+
"wandb_resume": null,
|
| 44 |
+
"wandb_mode": "online",
|
| 45 |
+
"no_wandb_sync": false,
|
| 46 |
+
"debug": false,
|
| 47 |
+
"resume_from": "outputs/healed/calib/glean_keep50_s1224/step0300",
|
| 48 |
+
"start_step": 0,
|
| 49 |
+
"no_grad_checkpointing": false,
|
| 50 |
+
"seed": 1224,
|
| 51 |
+
"no_teacher_overlap": false,
|
| 52 |
+
"sync_checkpoints": false,
|
| 53 |
+
"rollout_engine": "hf",
|
| 54 |
+
"vllm_gpu": null,
|
| 55 |
+
"vllm_port": 8377,
|
| 56 |
+
"vllm_refresh_every": 5,
|
| 57 |
+
"vllm_serve_bin": "vllm-plugin/.venv/bin/python",
|
| 58 |
+
"vllm_gpu_mem_util": 0.85,
|
| 59 |
+
"vllm_refresh_mode": "reload",
|
| 60 |
+
"vllm_live_dir": null,
|
| 61 |
+
"resolved_kl_direction": "forward"
|
| 62 |
+
}
|
healed/calib/glean_keep75_s1224.console.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/calib/glean_keep75_s1224.eval.log
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 1 |
38%|███▊ | 188/500 [00:00<00:00, 1879.71it/s]
|
| 2 |
76%|███████▋ | 382/500 [00:00<00:00, 1914.16it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 4 |
37%|███▋ | 187/500 [00:00<00:00, 1865.76it/s]
|
| 5 |
77%|███████▋ | 383/500 [00:00<00:00, 1920.11it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T21:35:29-07:00 serving outputs/healed/calib/glean_keep75_s1224/step0100 on GPU 2 port 8432
|
| 2 |
+
2026-07-17T21:35:29-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T21:35:59-07:00 server up; chat pass [gsm8k_cot_zeroshot]
|
| 4 |
+
2026-07-17:21:36:00 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 5 |
+
2026-07-17:21:36:08 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot']
|
| 6 |
+
2026-07-17:21:36:09 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 7 |
+
2026-07-17:21:36:09 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding!
|
| 8 |
+
2026-07-17:21:36:09 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8432/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 9 |
+
2026-07-17:21:36:09 INFO [models.api_models:179] Using max length 2048 - 1
|
| 10 |
+
2026-07-17:21:36:09 INFO [models.api_models:200] Using tokenizer None
|
| 11 |
+
2026-07-17:21:36:10 INFO [evaluator_utils:446] Selected tasks:
|
| 12 |
+
2026-07-17:21:36:10 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 13 |
+
2026-07-17:21:36:10 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280}
|
| 14 |
+
2026-07-17:21:36:10 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 15 |
+
|
| 16 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 17 |
38%|███▊ | 188/500 [00:00<00:00, 1879.71it/s]
|
| 18 |
76%|███████▋ | 382/500 [00:00<00:00, 1914.16it/s]
|
| 19 |
+
2026-07-17:21:36:11 INFO [evaluator:585] Running generate_until requests
|
| 20 |
+
2026-07-17:21:36:11 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 21 |
+
|
| 22 |
+
2026-07-17:21:37:06 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 23 |
+
2026-07-17:21:37:06 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/calib/keep75_step100/student/*.jsonl
|
| 24 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8432/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({'max_gen_toks': 1280}), limit: 500.0, num_fewshot: None, batch_size: 1
|
| 25 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 26 |
+
|------------------|------:|----------------|-----:|-----------|---|----:|---|-----:|
|
| 27 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match|↑ |0.652|± |0.0213|
|
| 28 |
+
| | |strict-match | 0|exact_match|↑ |0.000|± |0.0000|
|
| 29 |
+
|
| 30 |
+
2026-07-17T21:37:07-07:00 lm_eval exit=0 -> outputs/evals/general_suite/calib/keep75_step100
|
| 31 |
+
2026-07-17T23:22:47-07:00 serving outputs/healed/calib/glean_keep75_s1224/step0200 on GPU 2 port 8432
|
| 32 |
+
2026-07-17T23:22:47-07:00 waiting for server /health ...
|
| 33 |
+
2026-07-17T23:23:17-07:00 server up; chat pass [gsm8k_cot_zeroshot]
|
| 34 |
+
2026-07-17:23:23:17 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 35 |
+
2026-07-17:23:23:24 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot']
|
| 36 |
+
2026-07-17:23:23:25 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 37 |
+
2026-07-17:23:23:25 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding!
|
| 38 |
+
2026-07-17:23:23:25 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8432/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 39 |
+
2026-07-17:23:23:26 INFO [models.api_models:179] Using max length 2048 - 1
|
| 40 |
+
2026-07-17:23:23:26 INFO [models.api_models:200] Using tokenizer None
|
| 41 |
+
2026-07-17:23:23:27 INFO [evaluator_utils:446] Selected tasks:
|
| 42 |
+
2026-07-17:23:23:27 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 43 |
+
2026-07-17:23:23:27 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280}
|
| 44 |
+
2026-07-17:23:23:27 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 45 |
+
|
| 46 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 47 |
37%|███▋ | 187/500 [00:00<00:00, 1865.76it/s]
|
| 48 |
77%|███████▋ | 383/500 [00:00<00:00, 1920.11it/s]
|
| 49 |
+
2026-07-17:23:23:27 INFO [evaluator:585] Running generate_until requests
|
| 50 |
+
2026-07-17:23:23:27 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 51 |
+
|
| 52 |
+
2026-07-17:23:24:32 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 53 |
+
2026-07-17:23:24:32 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/calib/keep75_step200/student/*.jsonl
|
| 54 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8432/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({'max_gen_toks': 1280}), limit: 500.0, num_fewshot: None, batch_size: 1
|
| 55 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 56 |
+
|------------------|------:|----------------|-----:|-----------|---|----:|---|-----:|
|
| 57 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match|↑ |0.634|± |0.0216|
|
| 58 |
+
| | |strict-match | 0|exact_match|↑ |0.000|± |0.0000|
|
| 59 |
+
|
| 60 |
+
2026-07-17T23:24:33-07:00 lm_eval exit=0 -> outputs/evals/general_suite/calib/keep75_step200
|
healed/healing_breadth/glean_math_keep25_seed1224.console.log
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: setting up run zx83gucy
|
| 6 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 7 |
+
wandb: Run data is saved locally in outputs/healed/healing_breadth/glean_math_keep25_seed1224/wandb/run-20260714_205307-zx83gucy
|
| 8 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 9 |
+
wandb: Syncing run heal-glean_math_keep25-seed1224-run3
|
| 10 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal
|
| 11 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/zx83gucy
|
| 12 |
+
|
| 13 |
+
starting vllm rollout server on GPU GPU-864c54df-0130-7780-e271-8a5551d1733f (port 8377) ...
|
| 14 |
+
vllm server healthy in 24s
|
| 15 |
+
dataset_source filter ['omega', 'polaris', 'orz_math', 'mathsub', 'dapo-math'] -> 63998 prompts
|
| 16 |
+
|
| 17 |
+
difficulty <= 4 -> 63998 prompts
|
| 18 |
+
WARNING: difficulty filter removed nothing — the slice likely carries difficulty=None (Dolci math sources do), so the teacher-competence guard is NOT in effect
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
63977 prompts | 940 steps/epoch | 50 total steps | student params 2.09B | teacher overlap=True
|
| 22 |
+
rollout short by 2,838 tokens; generating top-up 1 from 3 prompts
|
| 23 |
+
{"step": 1, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.8929568549315134, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 47.5, "lr": 6e-06, "finish_rate": 0.216, "comp_len": 431.7, "t_data_s": 0.0, "t_rollout_s": 36.8, "t_step_s": 87.2, "t_refresh_s": 0.3, "mem_gb": 10.33}
|
| 24 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 25 |
+
[eval step 1] sample: '\nThe only number that could have been placed along thethe only number that is 75 divided by 5. The problem is to determine the maximum number of maames that could have been placed.\n\nThe only number th'
|
| 26 |
+
rollout short by 21,055 tokens; generating top-up 1 from 19 prompts
|
| 27 |
+
{"step": 2, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.8332483816136917, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 38.5, "lr": 9e-06, "finish_rate": 0.337, "comp_len": 371.5, "t_data_s": 0.0, "t_rollout_s": 38.3, "t_step_s": 87.5, "t_refresh_s": 0.3, "mem_gb": 10.45}
|
| 28 |
+
rollout short by 13,628 tokens; generating top-up 1 from 11 prompts
|
| 29 |
+
{"step": 3, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.7165187035838763, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 33.75, "lr": 1.2e-05, "finish_rate": 0.387, "comp_len": 393.4, "t_data_s": 0.0, "t_rollout_s": 37.4, "t_step_s": 84.6, "t_refresh_s": 0.3, "mem_gb": 10.35}
|
| 30 |
+
rollout short by 12,030 tokens; generating top-up 1 from 10 prompts
|
| 31 |
+
{"step": 4, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.5216001023660104, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 25.875, "lr": 1.5e-05, "finish_rate": 0.303, "comp_len": 394.7, "t_data_s": 0.0, "t_rollout_s": 36.1, "t_step_s": 80.7, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 32 |
+
rollout short by 16,574 tokens; generating top-up 1 from 14 prompts
|
| 33 |
+
{"step": 5, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.4106985497762758, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 29.875, "lr": 1.8e-05, "finish_rate": 0.429, "comp_len": 378.5, "t_data_s": 0.0, "t_rollout_s": 36.3, "t_step_s": 87.7, "t_refresh_s": 0.3, "mem_gb": 10.36}
|
| 34 |
+
rollout short by 14,735 tokens; generating top-up 1 from 12 prompts
|
| 35 |
+
{"step": 6, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.2116640572726727, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 28.625, "lr": 2.1e-05, "finish_rate": 0.55, "comp_len": 385.9, "t_data_s": 0.0, "t_rollout_s": 39.7, "t_step_s": 90.7, "t_refresh_s": 0.3, "mem_gb": 10.36}
|
| 36 |
+
rollout short by 9,665 tokens; generating top-up 1 from 8 prompts
|
| 37 |
+
{"step": 7, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 1.0176955454071364, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 12.9375, "lr": 2.4e-05, "finish_rate": 0.5, "comp_len": 402.7, "t_data_s": 0.0, "t_rollout_s": 39.5, "t_step_s": 85.6, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 38 |
+
{"step": 8, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.8163571097140511, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 7.1875, "lr": 2.7000000000000002e-05, "finish_rate": 0.317, "comp_len": 452.8, "t_data_s": 0.0, "t_rollout_s": 28.8, "t_step_s": 69.0, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 39 |
+
{"step": 9, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.7240555229172111, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 5.875, "lr": 3e-05, "finish_rate": 0.193, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 33.3, "t_step_s": 77.6, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 40 |
+
{"step": 10, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.691375209479034, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 3.734375, "lr": 3e-05, "finish_rate": 0.224, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 31.9, "t_step_s": 69.5, "t_refresh_s": 0.3, "mem_gb": 10.54}
|
| 41 |
+
[eval step 10] sample: "To solve this problem, we need to break it down into manageable steps and use logical reasoning. Let's start by analyzing the given conditions and the constraints.\n\n1. **Understanding the Problem:**\n "
|
| 42 |
+
{"step": 11, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6840026769215862, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 3.453125, "lr": 3e-05, "finish_rate": 0.181, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 33.9, "t_step_s": 70.7, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 43 |
+
{"step": 12, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6246620353197058, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 3.0625, "lr": 3e-05, "finish_rate": 0.165, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 34.2, "t_step_s": 71.5, "t_refresh_s": 0.3, "mem_gb": 10.52}
|
| 44 |
+
{"step": 13, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5813416921809316, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 2.328125, "lr": 3e-05, "finish_rate": 0.145, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 33.7, "t_step_s": 70.4, "t_refresh_s": 0.3, "mem_gb": 10.42}
|
| 45 |
+
{"step": 14, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.6156017889492214, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 3.828125, "lr": 3e-05, "finish_rate": 0.139, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 33.0, "t_step_s": 69.6, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 46 |
+
{"step": 15, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5399293902300298, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 2.265625, "lr": 3e-05, "finish_rate": 0.228, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 33.8, "t_step_s": 71.3, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 47 |
+
{"step": 16, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.5018366255449752, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 2.234375, "lr": 3e-05, "finish_rate": 0.2, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 33.8, "t_step_s": 71.5, "t_refresh_s": 0.3, "mem_gb": 10.44}
|
| 48 |
+
{"step": 17, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.49358773938765127, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 2.25, "lr": 3e-05, "finish_rate": 0.255, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 32.9, "t_step_s": 70.3, "t_refresh_s": 0.3, "mem_gb": 10.36}
|
| 49 |
+
{"step": 18, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.45969002382730445, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 2.203125, "lr": 3e-05, "finish_rate": 0.157, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 34.5, "t_step_s": 71.0, "t_refresh_s": 0.3, "mem_gb": 10.42}
|
| 50 |
+
{"step": 19, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4492309203994771, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 2.03125, "lr": 3e-05, "finish_rate": 0.192, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 34.1, "t_step_s": 71.6, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 51 |
+
{"step": 20, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4397894059561193, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 2.0, "lr": 3e-05, "finish_rate": 0.164, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 33.7, "t_step_s": 70.7, "t_refresh_s": 0.3, "mem_gb": 10.5}
|
| 52 |
+
[eval step 20] sample: "To solve this problem, we need to use combinatorial reasoning and constraints. Let's break it down step-by-step:\n\n1. **Define Variables:**\n - Let \\( m \\) be the number of mples.\n - Let \\( l \\) be "
|
| 53 |
+
{"step": 21, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.42920704178611435, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 1.890625, "lr": 3e-05, "finish_rate": 0.214, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 31.8, "t_step_s": 68.7, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 54 |
+
{"step": 22, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4285325840468208, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 1.8359375, "lr": 3e-05, "finish_rate": 0.209, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 34.4, "t_step_s": 71.7, "t_refresh_s": 0.3, "mem_gb": 10.42}
|
| 55 |
+
{"step": 23, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.39100426478857797, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 2.078125, "lr": 3e-05, "finish_rate": 0.117, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.1, "t_step_s": 71.4, "t_refresh_s": 0.3, "mem_gb": 10.43}
|
| 56 |
+
{"step": 24, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.44906475950367747, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 2.09375, "lr": 3e-05, "finish_rate": 0.172, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 33.4, "t_step_s": 71.4, "t_refresh_s": 0.3, "mem_gb": 10.51}
|
| 57 |
+
{"step": 25, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.4109231275471548, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 1.8125, "lr": 3e-05, "finish_rate": 0.25, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 32.5, "t_step_s": 70.7, "t_refresh_s": 0.3, "mem_gb": 10.53}
|
| 58 |
+
checkpoint snapshot queued -> outputs/healed/healing_breadth/glean_math_keep25_seed1224/step0025
|
| 59 |
+
{"step": 26, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3922373386658728, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 1.8125, "lr": 3e-05, "finish_rate": 0.251, "comp_len": 478.1, "t_data_s": 0.0, "t_rollout_s": 32.5, "t_step_s": 71.2, "t_refresh_s": 0.3, "mem_gb": 10.48}
|
| 60 |
+
{"step": 27, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.38296805831243597, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 1.859375, "lr": 3e-05, "finish_rate": 0.199, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 33.3, "t_step_s": 71.8, "t_refresh_s": 0.4, "mem_gb": 10.48}
|
| 61 |
+
{"step": 28, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.38206375452155866, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 1.6875, "lr": 3e-05, "finish_rate": 0.152, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 33.5, "t_step_s": 70.9, "t_refresh_s": 0.3, "mem_gb": 10.5}
|
| 62 |
+
{"step": 29, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3743179695730408, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 1.7421875, "lr": 3e-05, "finish_rate": 0.177, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 34.0, "t_step_s": 72.0, "t_refresh_s": 0.4, "mem_gb": 10.49}
|
| 63 |
+
{"step": 30, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.38618751456712685, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 1.796875, "lr": 3e-05, "finish_rate": 0.242, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 32.7, "t_step_s": 69.3, "t_refresh_s": 0.3, "mem_gb": 10.45}
|
| 64 |
+
[eval step 30] sample: 'To solve this problem, we need to determine the maximum number of maples that can be planted along the alley given the constraints:\n\n1. The total number of trees is 75.\n2. There are no two maples betw'
|
| 65 |
+
{"step": 31, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3496829746659845, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 1.6328125, "lr": 3e-05, "finish_rate": 0.232, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 30.5, "t_step_s": 68.1, "t_refresh_s": 0.3, "mem_gb": 10.45}
|
| 66 |
+
{"step": 32, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3457430052795758, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 1.7890625, "lr": 3e-05, "finish_rate": 0.194, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 32.5, "t_step_s": 69.8, "t_refresh_s": 0.3, "mem_gb": 10.48}
|
| 67 |
+
{"step": 33, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.36799976486029723, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 1.71875, "lr": 3e-05, "finish_rate": 0.232, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 31.6, "t_step_s": 69.4, "t_refresh_s": 0.3, "mem_gb": 10.44}
|
| 68 |
+
{"step": 34, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3638427472679565, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 1.6484375, "lr": 3e-05, "finish_rate": 0.257, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 31.5, "t_step_s": 69.5, "t_refresh_s": 0.3, "mem_gb": 10.51}
|
| 69 |
+
{"step": 35, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.36085621776568394, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 1.7265625, "lr": 3e-05, "finish_rate": 0.241, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 32.7, "t_step_s": 71.0, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 70 |
+
{"step": 36, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.327357999405389, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 1.5, "lr": 3e-05, "finish_rate": 0.237, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 32.7, "t_step_s": 71.1, "t_refresh_s": 0.3, "mem_gb": 10.5}
|
| 71 |
+
{"step": 37, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.34588671219460665, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 1.6875, "lr": 3e-05, "finish_rate": 0.171, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 32.9, "t_step_s": 70.7, "t_refresh_s": 0.3, "mem_gb": 10.42}
|
| 72 |
+
{"step": 38, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.34063462760510543, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 1.7109375, "lr": 3e-05, "finish_rate": 0.21, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 32.2, "t_step_s": 69.0, "t_refresh_s": 0.3, "mem_gb": 10.49}
|
| 73 |
+
{"step": 39, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.336651858022064, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 1.4921875, "lr": 3e-05, "finish_rate": 0.238, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 31.0, "t_step_s": 68.4, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 74 |
+
{"step": 40, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.32966975886846583, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 1.6875, "lr": 3e-05, "finish_rate": 0.246, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 32.7, "t_step_s": 70.5, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 75 |
+
[eval step 40] sample: "To solve this problem, we need to use combinatorial reasoning and constraints. Let's break down the problem step-by-step:\n\n1. **Define Variables:**\n - Let \\( m \\) be the number of maples.\n - Let \\"
|
| 76 |
+
{"step": 41, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.31904542534798386, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 1.703125, "lr": 3e-05, "finish_rate": 0.24, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 31.7, "t_step_s": 70.4, "t_refresh_s": 0.3, "mem_gb": 10.53}
|
| 77 |
+
{"step": 42, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3319877192765474, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 1.6796875, "lr": 3e-05, "finish_rate": 0.213, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 32.3, "t_step_s": 69.9, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 78 |
+
{"step": 43, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.32353154991058014, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 1.578125, "lr": 3e-05, "finish_rate": 0.293, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 32.2, "t_step_s": 69.6, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 79 |
+
{"step": 44, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3083366705554227, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 1.59375, "lr": 3e-05, "finish_rate": 0.217, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 31.9, "t_step_s": 69.0, "t_refresh_s": 0.3, "mem_gb": 10.44}
|
| 80 |
+
{"step": 45, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3119819214983533, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 1.59375, "lr": 3e-05, "finish_rate": 0.245, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 31.6, "t_step_s": 68.3, "t_refresh_s": 0.3, "mem_gb": 10.36}
|
| 81 |
+
{"step": 46, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.30937388323086, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 1.5, "lr": 3e-05, "finish_rate": 0.252, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 31.8, "t_step_s": 69.8, "t_refresh_s": 0.3, "mem_gb": 10.43}
|
| 82 |
+
{"step": 47, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2979762297540903, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 1.6171875, "lr": 3e-05, "finish_rate": 0.186, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 31.0, "t_step_s": 68.3, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 83 |
+
{"step": 48, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.31448883896246554, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 1.6015625, "lr": 3e-05, "finish_rate": 0.201, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 32.0, "t_step_s": 70.8, "t_refresh_s": 0.3, "mem_gb": 10.43}
|
| 84 |
+
{"step": 49, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2903304275934895, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 1.3671875, "lr": 3e-05, "finish_rate": 0.231, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 31.9, "t_step_s": 69.4, "t_refresh_s": 0.3, "mem_gb": 10.52}
|
| 85 |
+
{"step": 50, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.30557283139005303, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 1.5859375, "lr": 3e-05, "finish_rate": 0.25, "comp_len": 483.9, "t_data_s": 0.0, "t_rollout_s": 33.3, "t_step_s": 64.3, "t_refresh_s": 0.0, "mem_gb": 10.54}
|
| 86 |
+
[eval step 50] sample: "To solve this problem, we need to use combinatorial reasoning and constraints. Let's break down the problem step-by-step:\n\n1. **Define Variables:**\n - Let \\( m \\) be the number of maples.\n - Let \\"
|
| 87 |
+
checkpoint snapshot queued -> outputs/healed/healing_breadth/glean_math_keep25_seed1224/step0050
|
| 88 |
+
wandb: updating run metadata
|
| 89 |
+
wandb: uploading output.log; uploading wandb-summary.json; uploading config.yaml
|
| 90 |
+
wandb:
|
| 91 |
+
wandb: Run history:
|
| 92 |
+
wandb: comp_len ▄▁▂▂▁▃▅█▇███▇▇▇▇█▇██▇▇▇██▇▇▇▇▇▇▇▇▇▇▇▇▇▇▇
|
| 93 |
+
wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇██
|
| 94 |
+
wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 95 |
+
wandb: finish_rate ▃▅▆▄▇█▅▂▃▂▂▁▃▃▄▂▂▃▃▁▃▃▂▂▂▃▂▃▄▃▂▃▃▃▃▃▃▃▂▃
|
| 96 |
+
wandb: grad_norm █▇▆▅▅▃▂▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁��▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 97 |
+
wandb: lr ▁▂▃▄▅▆▇█████████████████████████████████
|
| 98 |
+
wandb: mem_gb ▁▅▂▂▂▂▂▃█▃▄▅▅▂▄▇▃▄▄▇▆▇▆▅▅▅▇▆▇▄▆▂█▅▃▂▄▅▄█
|
| 99 |
+
wandb: reverse_kl ██▇▆▆▄▃▃▃▂▂▂▂▂▂▂▂▂▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 100 |
+
wandb: step ▁▁▁▂▂▂▂▂▂▃▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇███
|
| 101 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 102 |
+
wandb: +4 ...
|
| 103 |
+
wandb:
|
| 104 |
+
wandb: Run summary:
|
| 105 |
+
wandb: comp_len 483.9
|
| 106 |
+
wandb: cumulative_loss_tokens 6000000
|
| 107 |
+
wandb: epoch 0
|
| 108 |
+
wandb: finish_rate 0.25
|
| 109 |
+
wandb: grad_norm 1.58594
|
| 110 |
+
wandb: lr 3e-05
|
| 111 |
+
wandb: mem_gb 10.54
|
| 112 |
+
wandb: reverse_kl 0.30557
|
| 113 |
+
wandb: step 50
|
| 114 |
+
wandb: t_data_s 0
|
| 115 |
+
wandb: +5 ...
|
| 116 |
+
wandb:
|
| 117 |
+
wandb: 🚀 View run heal-glean_math_keep25-seed1224-run3 at: https://wandb.ai/hbfreed/glean-heal/runs/zx83gucy
|
| 118 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal
|
| 119 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 120 |
+
wandb: Find logs at: outputs/healed/healing_breadth/glean_math_keep25_seed1224/wandb/run-20260714_205307-zx83gucy/logs
|
healed/healing_breadth/glean_math_keep25_seed1224_long.console.log
ADDED
|
@@ -0,0 +1,240 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: setting up run vpov0urn
|
| 6 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 7 |
+
wandb: Run data is saved locally in outputs/healed/healing_breadth/glean_math_keep25_seed1224_long/wandb/run-20260715_063332-vpov0urn
|
| 8 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 9 |
+
wandb: Syncing run heal-glean-math-keep25-seed1224-long-step500
|
| 10 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal
|
| 11 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/vpov0urn
|
| 12 |
+
|
| 13 |
+
resumed student weights from outputs/healed/healing_breadth/glean_math_keep25_seed1224/step0050 (fresh optimizer, step counter at 50)
|
| 14 |
+
starting vllm rollout server on GPU GPU-864c54df-0130-7780-e271-8a5551d1733f (port 8377) ...
|
| 15 |
+
vllm server healthy in 24s
|
| 16 |
+
dataset_source filter ['omega', 'polaris', 'orz_math', 'mathsub', 'dapo-math'] -> 63998 prompts
|
| 17 |
+
|
| 18 |
+
difficulty <= 4 -> 63998 prompts
|
| 19 |
+
WARNING: difficulty filter removed nothing — the slice likely carries difficulty=None (Dolci math sources do), so the teacher-competence guard is NOT in effect
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
63977 prompts | 940 steps/epoch | 500 total steps | student params 2.09B | teacher overlap=True
|
| 23 |
+
restored optimizer/scheduler state from step 50; rebuilt 228 paged buffers
|
| 24 |
+
{"step": 51, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.36993047905315957, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 2.203125, "lr": 3e-05, "finish_rate": 0.025, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 37.6, "t_step_s": 80.2, "t_refresh_s": 0.3, "mem_gb": 10.4}
|
| 25 |
+
wandb: updating run metadata
|
| 26 |
+
wandb: uploading summary
|
| 27 |
+
wandb:
|
| 28 |
+
wandb: Run history:
|
| 29 |
+
wandb: comp_len ▁
|
| 30 |
+
wandb: cumulative_loss_tokens ▁
|
| 31 |
+
wandb: epoch ▁
|
| 32 |
+
wandb: finish_rate ▁
|
| 33 |
+
wandb: grad_norm ▁
|
| 34 |
+
wandb: lr ▁
|
| 35 |
+
wandb: mem_gb ▁
|
| 36 |
+
wandb: reverse_kl ▁
|
| 37 |
+
wandb: step ▁
|
| 38 |
+
wandb: t_data_s ▁
|
| 39 |
+
wandb: +4 ...
|
| 40 |
+
wandb:
|
| 41 |
+
wandb: Run summary:
|
| 42 |
+
wandb: comp_len 508.5
|
| 43 |
+
wandb: cumulative_loss_tokens 6120000
|
| 44 |
+
wandb: epoch 0
|
| 45 |
+
wandb: finish_rate 0.025
|
| 46 |
+
wandb: grad_norm 2.20312
|
| 47 |
+
wandb: lr 3e-05
|
| 48 |
+
wandb: mem_gb 10.4
|
| 49 |
+
wandb: reverse_kl 0.36993
|
| 50 |
+
wandb: step 51
|
| 51 |
+
wandb: t_data_s 0
|
| 52 |
+
wandb: +5 ...
|
| 53 |
+
wandb:
|
| 54 |
+
wandb: 🚀 View run heal-glean-math-keep25-seed1224-long-step500 at: https://wandb.ai/hbfreed/glean-heal/runs/vpov0urn
|
| 55 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal
|
| 56 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 57 |
+
wandb: Find logs at: outputs/healed/healing_breadth/glean_math_keep25_seed1224_long/wandb/run-20260715_063332-vpov0urn/logs
|
| 58 |
+
Traceback (most recent call last):
|
| 59 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/scripts/11_distill_on_policy.py", line 1017, in <module>
|
| 60 |
+
main()
|
| 61 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/scripts/11_distill_on_policy.py", line 860, in main
|
| 62 |
+
save_student(student, tokenizer, live)
|
| 63 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/scripts/11_distill_on_policy.py", line 974, in save_student
|
| 64 |
+
write_student_snapshot(student, tokenizer, path, snapshot_student(student))
|
| 65 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/scripts/11_distill_on_policy.py", line 961, in write_student_snapshot
|
| 66 |
+
student.save_pretrained(path, state_dict=state_dict)
|
| 67 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/transformers/modeling_utils.py", line 4173, in save_pretrained
|
| 68 |
+
safe_save_file(shard, os.path.join(save_directory, shard_file), metadata=metadata)
|
| 69 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/safetensors/torch.py", line 323, in save_file
|
| 70 |
+
serialize_file(
|
| 71 |
+
safetensors._safetensors_rust.SafetensorError: Error while serializing: I/O error: No space left on device (os error 28)
|
| 72 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 73 |
+
warnings.warn('Grouped GEMM not available.')
|
| 74 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 75 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 76 |
+
wandb: setting up run i5fg3xzq
|
| 77 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 78 |
+
wandb: Run data is saved locally in outputs/healed/healing_breadth/glean_math_keep25_seed1224_long/wandb/run-20260715_071701-i5fg3xzq
|
| 79 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 80 |
+
wandb: Syncing run heal-glean-math-keep25-seed1224-long-step500
|
| 81 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal
|
| 82 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/i5fg3xzq
|
| 83 |
+
|
| 84 |
+
resumed student weights from outputs/healed/healing_breadth/glean_math_keep25_seed1224/step0050 (fresh optimizer, step counter at 50)
|
| 85 |
+
starting vllm rollout server on GPU GPU-864c54df-0130-7780-e271-8a5551d1733f (port 8377) ...
|
| 86 |
+
vllm server healthy in 24s
|
| 87 |
+
dataset_source filter ['omega', 'polaris', 'orz_math', 'mathsub', 'dapo-math'] -> 63998 prompts
|
| 88 |
+
difficulty <= 4 -> 63998 prompts
|
| 89 |
+
WARNING: difficulty filter removed nothing — the slice likely carries difficulty=None (Dolci math sources do), so the teacher-competence guard is NOT in effect
|
| 90 |
+
63977 prompts | 940 steps/epoch | 500 total steps | student params 2.09B | teacher overlap=True
|
| 91 |
+
restored optimizer/scheduler state from step 50; rebuilt 228 paged buffers
|
| 92 |
+
{"step": 51, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.36993047905315957, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 2.203125, "lr": 3e-05, "finish_rate": 0.025, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 37.4, "t_step_s": 79.7, "t_refresh_s": 0.4, "mem_gb": 10.4}
|
| 93 |
+
{"step": 52, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3874280411116779, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 1.765625, "lr": 3e-05, "finish_rate": 0.021, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 37.0, "t_step_s": 72.9, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 94 |
+
{"step": 53, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3746817006888489, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 1.703125, "lr": 3e-05, "finish_rate": 0.046, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 37.1, "t_step_s": 73.6, "t_refresh_s": 0.3, "mem_gb": 10.45}
|
| 95 |
+
{"step": 54, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.35510130622684954, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 2.140625, "lr": 3e-05, "finish_rate": 0.021, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 37.7, "t_step_s": 74.6, "t_refresh_s": 0.3, "mem_gb": 10.49}
|
| 96 |
+
{"step": 55, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.34786954234316947, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 1.8359375, "lr": 3e-05, "finish_rate": 0.025, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 36.9, "t_step_s": 72.5, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 97 |
+
{"step": 56, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3465809899068127, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 1.6796875, "lr": 3e-05, "finish_rate": 0.038, "comp_len": 508.5, "t_data_s": 0.0, "t_rollout_s": 36.9, "t_step_s": 72.6, "t_refresh_s": 0.3, "mem_gb": 10.43}
|
| 98 |
+
{"step": 57, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.37013316290453074, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 2.34375, "lr": 3e-05, "finish_rate": 0.059, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 36.9, "t_step_s": 72.9, "t_refresh_s": 0.3, "mem_gb": 10.51}
|
| 99 |
+
{"step": 58, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3522711561133464, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 1.6328125, "lr": 3e-05, "finish_rate": 0.104, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.5, "t_step_s": 72.8, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 100 |
+
{"step": 59, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3397159261740744, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 1.8203125, "lr": 3e-05, "finish_rate": 0.055, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 36.8, "t_step_s": 72.5, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 101 |
+
{"step": 60, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3382764852608244, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 1.7265625, "lr": 3e-05, "finish_rate": 0.08, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 36.7, "t_step_s": 72.7, "t_refresh_s": 0.3, "mem_gb": 10.42}
|
| 102 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 103 |
+
[eval step 60] sample: "To solve this problem, we need to consider the constraints and the total number of trees. Let's denote:\n- \\( m \\) as the number of maples.\n- \\( l \\) as the number of larks.\n\nGiven:\n1. The total number"
|
| 104 |
+
{"step": 60, "gsm8k_n": 64, "gsm8k_quick_chat": 0.390625, "t_eval_s": 46.9}
|
| 105 |
+
{"step": 61, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.31868271032124756, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 1.5859375, "lr": 3e-05, "finish_rate": 0.059, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 37.5, "t_step_s": 74.1, "t_refresh_s": 0.3, "mem_gb": 12.44}
|
| 106 |
+
{"step": 62, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3191785570286214, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 1.5, "lr": 3e-05, "finish_rate": 0.063, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 36.9, "t_step_s": 73.1, "t_refresh_s": 0.3, "mem_gb": 10.43}
|
| 107 |
+
{"step": 63, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3180096957164506, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 1.5625, "lr": 3e-05, "finish_rate": 0.092, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 35.3, "t_step_s": 70.9, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 108 |
+
{"step": 64, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.288780878906697, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 1.7578125, "lr": 3e-05, "finish_rate": 0.046, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 37.6, "t_step_s": 74.6, "t_refresh_s": 0.3, "mem_gb": 10.51}
|
| 109 |
+
{"step": 65, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.29233540159290033, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 1.4921875, "lr": 3e-05, "finish_rate": 0.084, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 36.2, "t_step_s": 71.9, "t_refresh_s": 0.3, "mem_gb": 10.5}
|
| 110 |
+
{"step": 66, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2769135721528282, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 1.3515625, "lr": 3e-05, "finish_rate": 0.105, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 35.6, "t_step_s": 71.3, "t_refresh_s": 0.3, "mem_gb": 10.53}
|
| 111 |
+
{"step": 67, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3056252569064498, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 1.4375, "lr": 3e-05, "finish_rate": 0.181, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 33.5, "t_step_s": 69.5, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 112 |
+
{"step": 68, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.30779550124146043, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 1.5, "lr": 3e-05, "finish_rate": 0.129, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.9, "t_step_s": 71.2, "t_refresh_s": 0.3, "mem_gb": 10.42}
|
| 113 |
+
{"step": 69, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2875990556370467, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 1.4609375, "lr": 3e-05, "finish_rate": 0.132, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 33.8, "t_step_s": 69.7, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 114 |
+
{"step": 70, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.28901881557305653, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 1.3828125, "lr": 3e-05, "finish_rate": 0.108, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 36.2, "t_step_s": 73.7, "t_refresh_s": 0.3, "mem_gb": 10.54}
|
| 115 |
+
[eval step 70] sample: 'To solve this problem, we need to determine the maximum number of maples that can be planted along the alley given the constraints:\n\n1. The total number of trees is 75.\n2. There are no two maples betw'
|
| 116 |
+
{"step": 70, "gsm8k_n": 64, "gsm8k_quick_chat": 0.28125, "t_eval_s": 44.7}
|
| 117 |
+
{"step": 71, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.281591778382659, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 1.53125, "lr": 3e-05, "finish_rate": 0.144, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 33.2, "t_step_s": 69.5, "t_refresh_s": 0.3, "mem_gb": 12.44}
|
| 118 |
+
{"step": 72, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.27427094464078544, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 1.4375, "lr": 3e-05, "finish_rate": 0.107, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 35.8, "t_step_s": 73.1, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 119 |
+
{"step": 73, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.25415447360488275, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 1.3359375, "lr": 3e-05, "finish_rate": 0.116, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.4, "t_step_s": 71.6, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 120 |
+
{"step": 74, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2598834279651443, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 1.1875, "lr": 3e-05, "finish_rate": 0.109, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 35.8, "t_step_s": 72.9, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 121 |
+
{"step": 75, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.3036417054095616, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 1.40625, "lr": 3e-05, "finish_rate": 0.133, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.9, "t_step_s": 71.1, "t_refresh_s": 0.3, "mem_gb": 10.41}
|
| 122 |
+
{"step": 76, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.28468160825570427, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 1.40625, "lr": 3e-05, "finish_rate": 0.124, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.7, "t_step_s": 71.7, "t_refresh_s": 0.3, "mem_gb": 10.45}
|
| 123 |
+
{"step": 77, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.26907755986650783, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 1.2109375, "lr": 3e-05, "finish_rate": 0.137, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.0, "t_step_s": 70.4, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 124 |
+
{"step": 78, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23572471514294546, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 1.3125, "lr": 3e-05, "finish_rate": 0.113, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 36.0, "t_step_s": 72.0, "t_refresh_s": 0.3, "mem_gb": 10.45}
|
| 125 |
+
{"step": 79, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2453982294137279, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 1.296875, "lr": 3e-05, "finish_rate": 0.156, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 33.0, "t_step_s": 69.6, "t_refresh_s": 0.3, "mem_gb": 10.43}
|
| 126 |
+
{"step": 80, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2668868999974181, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 2.328125, "lr": 3e-05, "finish_rate": 0.149, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.2, "t_step_s": 70.1, "t_refresh_s": 0.3, "mem_gb": 10.4}
|
| 127 |
+
[eval step 80] sample: 'To solve this problem, we need to determine the maximum number of maples that can be planted along the alley given the constraints:\n\n1. The total number of trees is 75.\n2. There are no two maples betw'
|
| 128 |
+
{"step": 80, "gsm8k_n": 64, "gsm8k_quick_chat": 0.390625, "t_eval_s": 44.8}
|
| 129 |
+
{"step": 81, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.27245304357651623, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 1.46875, "lr": 3e-05, "finish_rate": 0.097, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 35.7, "t_step_s": 71.9, "t_refresh_s": 0.3, "mem_gb": 12.44}
|
| 130 |
+
{"step": 82, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.26285181757311027, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 1.1875, "lr": 3e-05, "finish_rate": 0.117, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.5, "t_step_s": 71.6, "t_refresh_s": 0.3, "mem_gb": 10.42}
|
| 131 |
+
{"step": 83, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2504500327867145, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 1.390625, "lr": 3e-05, "finish_rate": 0.157, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 34.4, "t_step_s": 71.2, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 132 |
+
{"step": 84, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.25419983249952394, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 1.640625, "lr": 3e-05, "finish_rate": 0.165, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 33.9, "t_step_s": 70.4, "t_refresh_s": 0.3, "mem_gb": 10.43}
|
| 133 |
+
{"step": 85, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.24969746678496402, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 1.3203125, "lr": 3e-05, "finish_rate": 0.117, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 35.8, "t_step_s": 73.1, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 134 |
+
{"step": 86, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.26418633902855215, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 1.3984375, "lr": 3e-05, "finish_rate": 0.172, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 34.2, "t_step_s": 71.4, "t_refresh_s": 0.3, "mem_gb": 10.48}
|
| 135 |
+
{"step": 87, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.25853364124981065, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 1.3515625, "lr": 3e-05, "finish_rate": 0.137, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.3, "t_step_s": 70.2, "t_refresh_s": 0.3, "mem_gb": 10.4}
|
| 136 |
+
{"step": 88, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.26424587182166676, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 1.328125, "lr": 3e-05, "finish_rate": 0.132, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 34.2, "t_step_s": 70.9, "t_refresh_s": 0.3, "mem_gb": 10.48}
|
| 137 |
+
{"step": 89, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.247395279224962, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 1.2109375, "lr": 3e-05, "finish_rate": 0.12, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.5, "t_step_s": 70.4, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 138 |
+
{"step": 90, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.27036167939615746, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 1.3359375, "lr": 3e-05, "finish_rate": 0.112, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 34.6, "t_step_s": 71.2, "t_refresh_s": 0.3, "mem_gb": 10.53}
|
| 139 |
+
[eval step 90] sample: "To solve this problem, we need to maximize the number of maples \\( M \\) such that the total number of trees \\( T \\) is 75, and there are no two maples between which there are exactly 5 trees.\n\nLet's b"
|
| 140 |
+
{"step": 90, "gsm8k_n": 64, "gsm8k_quick_chat": 0.359375, "t_eval_s": 44.7}
|
| 141 |
+
{"step": 91, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2520888489575436, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 1.203125, "lr": 3e-05, "finish_rate": 0.216, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 33.6, "t_step_s": 70.4, "t_refresh_s": 0.3, "mem_gb": 12.44}
|
| 142 |
+
{"step": 92, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23066797492839397, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 1.2421875, "lr": 3e-05, "finish_rate": 0.088, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 34.6, "t_step_s": 70.4, "t_refresh_s": 0.3, "mem_gb": 10.43}
|
| 143 |
+
{"step": 93, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23549754046387972, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 1.3125, "lr": 3e-05, "finish_rate": 0.088, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 37.7, "t_step_s": 74.4, "t_refresh_s": 0.3, "mem_gb": 10.54}
|
| 144 |
+
{"step": 94, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2587053025153776, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 1.40625, "lr": 3e-05, "finish_rate": 0.126, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 35.1, "t_step_s": 71.5, "t_refresh_s": 0.3, "mem_gb": 10.45}
|
| 145 |
+
{"step": 95, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.25772884955219927, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 1.2265625, "lr": 3e-05, "finish_rate": 0.059, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 37.7, "t_step_s": 74.7, "t_refresh_s": 0.3, "mem_gb": 10.52}
|
| 146 |
+
{"step": 96, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2583792527654519, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 1.296875, "lr": 3e-05, "finish_rate": 0.145, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 33.8, "t_step_s": 70.6, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 147 |
+
{"step": 97, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2516133760511875, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 1.3984375, "lr": 3e-05, "finish_rate": 0.174, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 33.8, "t_step_s": 71.0, "t_refresh_s": 0.3, "mem_gb": 10.56}
|
| 148 |
+
{"step": 98, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2511387197598815, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 1.25, "lr": 3e-05, "finish_rate": 0.083, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 36.0, "t_step_s": 72.5, "t_refresh_s": 0.3, "mem_gb": 10.5}
|
| 149 |
+
{"step": 99, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.25845928863280765, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 1.328125, "lr": 3e-05, "finish_rate": 0.121, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.0, "t_step_s": 71.0, "t_refresh_s": 0.3, "mem_gb": 10.4}
|
| 150 |
+
{"step": 100, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2377076770828416, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 1.203125, "lr": 3e-05, "finish_rate": 0.177, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 32.8, "t_step_s": 69.4, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 151 |
+
[eval step 100] sample: 'To solve this problem, we need to maximize the number of maples (M) that can be planted along the alley such that the total number of trees (T) is 75, and there are no two maples between which there a'
|
| 152 |
+
{"step": 100, "gsm8k_n": 64, "gsm8k_quick_chat": 0.328125, "t_eval_s": 44.8}
|
| 153 |
+
checkpoint snapshot queued -> outputs/healed/healing_breadth/glean_math_keep25_seed1224_long/step0100
|
| 154 |
+
{"step": 101, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2684837412368506, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 1.296875, "lr": 3e-05, "finish_rate": 0.196, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 33.0, "t_step_s": 69.4, "t_refresh_s": 0.3, "mem_gb": 12.44}
|
| 155 |
+
{"step": 102, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.24094757018235202, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 1.3046875, "lr": 3e-05, "finish_rate": 0.141, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.0, "t_step_s": 70.5, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 156 |
+
{"step": 103, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2634418107945472, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 1.2265625, "lr": 3e-05, "finish_rate": 0.181, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 33.4, "t_step_s": 70.0, "t_refresh_s": 0.3, "mem_gb": 10.4}
|
| 157 |
+
{"step": 104, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22899044912308456, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 1.171875, "lr": 3e-05, "finish_rate": 0.136, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 33.8, "t_step_s": 70.1, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 158 |
+
{"step": 105, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23377793765577176, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 1.203125, "lr": 3e-05, "finish_rate": 0.128, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 34.9, "t_step_s": 72.2, "t_refresh_s": 0.3, "mem_gb": 10.52}
|
| 159 |
+
{"step": 106, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23661676298119128, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 1.1484375, "lr": 3e-05, "finish_rate": 0.121, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.6, "t_step_s": 72.4, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 160 |
+
{"step": 107, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22338223065932591, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 1.1328125, "lr": 3e-05, "finish_rate": 0.145, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 34.3, "t_step_s": 71.2, "t_refresh_s": 0.3, "mem_gb": 10.49}
|
| 161 |
+
{"step": 108, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.24060181738672157, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 1.34375, "lr": 3e-05, "finish_rate": 0.168, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 33.4, "t_step_s": 70.5, "t_refresh_s": 0.3, "mem_gb": 10.54}
|
| 162 |
+
{"step": 109, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22999099592032532, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 1.1796875, "lr": 3e-05, "finish_rate": 0.161, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 33.9, "t_step_s": 70.3, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 163 |
+
{"step": 110, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.24095743914954365, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 1.578125, "lr": 3e-05, "finish_rate": 0.112, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 35.2, "t_step_s": 71.6, "t_refresh_s": 0.3, "mem_gb": 10.48}
|
| 164 |
+
[eval step 110] sample: 'To solve this problem, we need to determine the maximum number of maples (\\(M\\)) that can be planted along the alley such that the total number of trees (\\(T\\)) is 75, and there are no two maples betw'
|
| 165 |
+
{"step": 110, "gsm8k_n": 64, "gsm8k_quick_chat": 0.390625, "t_eval_s": 45.3}
|
| 166 |
+
{"step": 111, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21752187181736032, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 1.0859375, "lr": 3e-05, "finish_rate": 0.217, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 30.9, "t_step_s": 68.1, "t_refresh_s": 0.3, "mem_gb": 12.44}
|
| 167 |
+
{"step": 112, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2314429316678395, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 1.3359375, "lr": 3e-05, "finish_rate": 0.12, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.8, "t_step_s": 71.5, "t_refresh_s": 0.3, "mem_gb": 10.52}
|
| 168 |
+
{"step": 113, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2235619456699739, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 1.1484375, "lr": 3e-05, "finish_rate": 0.137, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 33.9, "t_step_s": 71.1, "t_refresh_s": 0.3, "mem_gb": 10.55}
|
| 169 |
+
{"step": 114, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23131392294329903, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 1.234375, "lr": 3e-05, "finish_rate": 0.124, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 34.3, "t_step_s": 71.2, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 170 |
+
{"step": 115, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22863815203296642, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 1.2578125, "lr": 3e-05, "finish_rate": 0.104, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.9, "t_step_s": 71.7, "t_refresh_s": 0.3, "mem_gb": 10.4}
|
| 171 |
+
{"step": 116, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22345312169020376, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 1.3203125, "lr": 3e-05, "finish_rate": 0.125, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.2, "t_step_s": 72.0, "t_refresh_s": 0.3, "mem_gb": 10.48}
|
| 172 |
+
{"step": 117, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22978670983004074, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 1.1328125, "lr": 3e-05, "finish_rate": 0.108, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 35.8, "t_step_s": 73.0, "t_refresh_s": 0.3, "mem_gb": 10.52}
|
| 173 |
+
{"step": 118, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2331024925402055, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 1.28125, "lr": 3e-05, "finish_rate": 0.076, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 37.3, "t_step_s": 73.9, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 174 |
+
{"step": 119, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22393636475646247, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 1.234375, "lr": 3e-05, "finish_rate": 0.13, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 35.4, "t_step_s": 71.5, "t_refresh_s": 0.3, "mem_gb": 10.44}
|
| 175 |
+
{"step": 120, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2355032395929719, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 1.203125, "lr": 3e-05, "finish_rate": 0.16, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 33.1, "t_step_s": 68.8, "t_refresh_s": 0.3, "mem_gb": 10.36}
|
| 176 |
+
[eval step 120] sample: 'To solve this problem, we need to determine the maximum number of maples that can be planted along the alley given the constraints:\n\n1. The total number of trees is 75.\n2. There are no two maples betw'
|
| 177 |
+
{"step": 120, "gsm8k_n": 64, "gsm8k_quick_chat": 0.34375, "t_eval_s": 44.9}
|
| 178 |
+
{"step": 121, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.24267969185064237, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 1.171875, "lr": 3e-05, "finish_rate": 0.113, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 34.9, "t_step_s": 71.1, "t_refresh_s": 0.3, "mem_gb": 12.43}
|
| 179 |
+
{"step": 122, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23425059191025793, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 1.1640625, "lr": 3e-05, "finish_rate": 0.093, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 37.0, "t_step_s": 74.2, "t_refresh_s": 0.3, "mem_gb": 10.53}
|
| 180 |
+
{"step": 123, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.26844764651320874, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 1.359375, "lr": 3e-05, "finish_rate": 0.096, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.0, "t_step_s": 71.1, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 181 |
+
{"step": 124, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21723629867484173, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 1.2578125, "lr": 3e-05, "finish_rate": 0.116, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 33.6, "t_step_s": 69.0, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 182 |
+
{"step": 125, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21943003199926267, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 1.3515625, "lr": 3e-05, "finish_rate": 0.14, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 34.5, "t_step_s": 70.8, "t_refresh_s": 0.3, "mem_gb": 10.49}
|
| 183 |
+
{"step": 126, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.214982470301042, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 1.171875, "lr": 3e-05, "finish_rate": 0.13, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 34.9, "t_step_s": 70.3, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 184 |
+
{"step": 127, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22647308003318806, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 1.34375, "lr": 3e-05, "finish_rate": 0.105, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 35.0, "t_step_s": 71.0, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 185 |
+
{"step": 128, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23893264475651085, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 1.4140625, "lr": 3e-05, "finish_rate": 0.104, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.3, "t_step_s": 71.4, "t_refresh_s": 0.3, "mem_gb": 10.5}
|
| 186 |
+
{"step": 129, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23907007528046767, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 1.1796875, "lr": 3e-05, "finish_rate": 0.127, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 31.7, "t_step_s": 69.0, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 187 |
+
{"step": 130, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23370486757767697, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 1.2421875, "lr": 3e-05, "finish_rate": 0.139, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 32.2, "t_step_s": 68.9, "t_refresh_s": 0.3, "mem_gb": 10.43}
|
| 188 |
+
[eval step 130] sample: 'To solve this problem, we need to determine the maximum number of maples that can be planted along the alley given the constraints:\n\n1. The total number of trees is 75.\n2. There are no two maples betw'
|
| 189 |
+
{"step": 130, "gsm8k_n": 64, "gsm8k_quick_chat": 0.375, "t_eval_s": 45.3}
|
| 190 |
+
{"step": 131, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22123020526878537, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 1.2421875, "lr": 3e-05, "finish_rate": 0.088, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 36.5, "t_step_s": 73.1, "t_refresh_s": 0.3, "mem_gb": 12.44}
|
| 191 |
+
{"step": 132, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22262640608406314, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 1.203125, "lr": 3e-05, "finish_rate": 0.129, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 33.3, "t_step_s": 69.3, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 192 |
+
{"step": 133, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23642245450820773, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 1.3125, "lr": 3e-05, "finish_rate": 0.141, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 33.9, "t_step_s": 70.1, "t_refresh_s": 0.3, "mem_gb": 10.5}
|
| 193 |
+
{"step": 134, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22181539568305014, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 1.1484375, "lr": 3e-05, "finish_rate": 0.124, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 35.1, "t_step_s": 71.9, "t_refresh_s": 0.3, "mem_gb": 10.49}
|
| 194 |
+
{"step": 135, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21886845049690457, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 1.265625, "lr": 3e-05, "finish_rate": 0.088, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 35.4, "t_step_s": 71.3, "t_refresh_s": 0.3, "mem_gb": 10.43}
|
| 195 |
+
{"step": 136, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23528965907134117, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 1.2421875, "lr": 3e-05, "finish_rate": 0.117, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.0, "t_step_s": 71.1, "t_refresh_s": 0.3, "mem_gb": 10.45}
|
| 196 |
+
{"step": 137, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22028850772418082, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 1.3828125, "lr": 3e-05, "finish_rate": 0.076, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 36.0, "t_step_s": 71.8, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 197 |
+
{"step": 138, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22388356751191119, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 1.171875, "lr": 3e-05, "finish_rate": 0.123, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 33.4, "t_step_s": 69.5, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 198 |
+
{"step": 139, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2186046968769282, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 1.1171875, "lr": 3e-05, "finish_rate": 0.084, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 36.2, "t_step_s": 72.8, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 199 |
+
{"step": 140, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21518618610985576, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 1.1171875, "lr": 3e-05, "finish_rate": 0.121, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 33.9, "t_step_s": 69.8, "t_refresh_s": 0.3, "mem_gb": 10.44}
|
| 200 |
+
[eval step 140] sample: 'To solve this problem, we need to carefully analyze the constraints given:\n\n1. **Total number of trees:** 75 trees.\n2. **No two maples between which there are exactly 5 trees:** This means that no two'
|
| 201 |
+
{"step": 140, "gsm8k_n": 64, "gsm8k_quick_chat": 0.359375, "t_eval_s": 44.8}
|
| 202 |
+
{"step": 141, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21726824556073795, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 1.1953125, "lr": 3e-05, "finish_rate": 0.072, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 37.1, "t_step_s": 73.1, "t_refresh_s": 0.3, "mem_gb": 12.43}
|
| 203 |
+
{"step": 142, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21776566527485847, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 1.2109375, "lr": 3e-05, "finish_rate": 0.104, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 33.9, "t_step_s": 70.1, "t_refresh_s": 0.3, "mem_gb": 10.4}
|
| 204 |
+
{"step": 143, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21226181217512738, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 1.234375, "lr": 3e-05, "finish_rate": 0.128, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 33.6, "t_step_s": 70.3, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 205 |
+
{"step": 144, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.20745828753014406, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 1.09375, "lr": 3e-05, "finish_rate": 0.141, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 35.2, "t_step_s": 71.5, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 206 |
+
{"step": 145, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.20095169744286687, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 1.1796875, "lr": 3e-05, "finish_rate": 0.067, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 37.0, "t_step_s": 72.8, "t_refresh_s": 0.3, "mem_gb": 10.5}
|
| 207 |
+
{"step": 146, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21430944070120653, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 1.15625, "lr": 3e-05, "finish_rate": 0.092, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 35.2, "t_step_s": 71.4, "t_refresh_s": 0.3, "mem_gb": 10.41}
|
| 208 |
+
{"step": 147, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.20074411552796761, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 1.1015625, "lr": 3e-05, "finish_rate": 0.092, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 35.3, "t_step_s": 71.3, "t_refresh_s": 0.3, "mem_gb": 10.51}
|
| 209 |
+
{"step": 148, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2208368306187292, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 1.1328125, "lr": 3e-05, "finish_rate": 0.148, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 33.4, "t_step_s": 69.9, "t_refresh_s": 0.3, "mem_gb": 10.4}
|
| 210 |
+
{"step": 149, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22132619506145518, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 1.265625, "lr": 3e-05, "finish_rate": 0.108, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 36.8, "t_step_s": 74.1, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 211 |
+
{"step": 150, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2195085643796871, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 1.1328125, "lr": 3e-05, "finish_rate": 0.112, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 36.1, "t_step_s": 73.3, "t_refresh_s": 0.3, "mem_gb": 10.53}
|
| 212 |
+
[eval step 150] sample: "To solve this problem, we need to determine the maximum number of maples that can be placed along the alley such that there are no two maples between which there are exactly 5 trees.\n\nLet's break down"
|
| 213 |
+
{"step": 150, "gsm8k_n": 64, "gsm8k_quick_chat": 0.3125, "t_eval_s": 44.9}
|
| 214 |
+
checkpoint snapshot queued -> outputs/healed/healing_breadth/glean_math_keep25_seed1224_long/step0150
|
| 215 |
+
{"step": 151, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21052663061742982, "tokens": 120000, "cumulative_loss_tokens": 18120000, "grad_norm": 1.234375, "lr": 3e-05, "finish_rate": 0.1, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 35.0, "t_step_s": 70.8, "t_refresh_s": 0.3, "mem_gb": 12.43}
|
| 216 |
+
{"step": 152, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.23573231500076752, "tokens": 120000, "cumulative_loss_tokens": 18240000, "grad_norm": 1.140625, "lr": 3e-05, "finish_rate": 0.076, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 37.6, "t_step_s": 74.7, "t_refresh_s": 0.3, "mem_gb": 10.54}
|
| 217 |
+
{"step": 153, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21262744003695747, "tokens": 120000, "cumulative_loss_tokens": 18360000, "grad_norm": 1.125, "lr": 3e-05, "finish_rate": 0.152, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 33.7, "t_step_s": 70.2, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 218 |
+
{"step": 154, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.20737704386965683, "tokens": 120000, "cumulative_loss_tokens": 18480000, "grad_norm": 1.125, "lr": 3e-05, "finish_rate": 0.155, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 32.1, "t_step_s": 69.1, "t_refresh_s": 0.3, "mem_gb": 10.4}
|
| 219 |
+
{"step": 155, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.20515705520591387, "tokens": 120000, "cumulative_loss_tokens": 18600000, "grad_norm": 1.125, "lr": 3e-05, "finish_rate": 0.125, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 34.8, "t_step_s": 72.9, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 220 |
+
{"step": 156, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.227416008350874, "tokens": 120000, "cumulative_loss_tokens": 18720000, "grad_norm": 1.1640625, "lr": 3e-05, "finish_rate": 0.143, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 33.7, "t_step_s": 70.4, "t_refresh_s": 0.3, "mem_gb": 10.46}
|
| 221 |
+
{"step": 157, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2179569326767077, "tokens": 120000, "cumulative_loss_tokens": 18840000, "grad_norm": 1.1953125, "lr": 3e-05, "finish_rate": 0.116, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.7, "t_step_s": 70.6, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 222 |
+
{"step": 158, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21790637281499803, "tokens": 120000, "cumulative_loss_tokens": 18960000, "grad_norm": 1.1953125, "lr": 3e-05, "finish_rate": 0.12, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 33.5, "t_step_s": 69.6, "t_refresh_s": 0.3, "mem_gb": 10.37}
|
| 223 |
+
{"step": 159, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21037111974650374, "tokens": 120000, "cumulative_loss_tokens": 19080000, "grad_norm": 1.25, "lr": 3e-05, "finish_rate": 0.124, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.3, "t_step_s": 70.5, "t_refresh_s": 0.3, "mem_gb": 10.39}
|
| 224 |
+
{"step": 160, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22135015249016385, "tokens": 120000, "cumulative_loss_tokens": 19200000, "grad_norm": 1.234375, "lr": 3e-05, "finish_rate": 0.1, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 35.2, "t_step_s": 72.0, "t_refresh_s": 0.3, "mem_gb": 10.44}
|
| 225 |
+
[eval step 160] sample: "To solve this problem, we need to determine the maximum number of maples that can be placed along the alley such that there are no two maples between which there are exactly 5 trees.\n\nLet's break down"
|
| 226 |
+
{"step": 160, "gsm8k_n": 64, "gsm8k_quick_chat": 0.375, "t_eval_s": 44.8}
|
| 227 |
+
{"step": 161, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.19312054982806245, "tokens": 120000, "cumulative_loss_tokens": 19320000, "grad_norm": 1.078125, "lr": 3e-05, "finish_rate": 0.076, "comp_len": 506.3, "t_data_s": 0.0, "t_rollout_s": 37.0, "t_step_s": 73.0, "t_refresh_s": 0.3, "mem_gb": 12.43}
|
| 228 |
+
{"step": 162, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21525976048012574, "tokens": 120000, "cumulative_loss_tokens": 19440000, "grad_norm": 1.2734375, "lr": 3e-05, "finish_rate": 0.177, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 33.4, "t_step_s": 69.6, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
| 229 |
+
{"step": 163, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21099251879341901, "tokens": 120000, "cumulative_loss_tokens": 19560000, "grad_norm": 1.171875, "lr": 3e-05, "finish_rate": 0.108, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.3, "t_step_s": 71.9, "t_refresh_s": 0.3, "mem_gb": 10.41}
|
| 230 |
+
{"step": 164, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2188319661433498, "tokens": 120000, "cumulative_loss_tokens": 19680000, "grad_norm": 1.1953125, "lr": 3e-05, "finish_rate": 0.136, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 33.7, "t_step_s": 70.1, "t_refresh_s": 0.3, "mem_gb": 10.47}
|
| 231 |
+
{"step": 165, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21699987841347854, "tokens": 120000, "cumulative_loss_tokens": 19800000, "grad_norm": 1.1640625, "lr": 3e-05, "finish_rate": 0.137, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.7, "t_step_s": 72.0, "t_refresh_s": 0.3, "mem_gb": 10.54}
|
| 232 |
+
{"step": 166, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.20422210597346227, "tokens": 120000, "cumulative_loss_tokens": 19920000, "grad_norm": 1.1875, "lr": 3e-05, "finish_rate": 0.088, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 34.4, "t_step_s": 69.9, "t_refresh_s": 0.3, "mem_gb": 10.36}
|
| 233 |
+
{"step": 167, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.21616086491048336, "tokens": 120000, "cumulative_loss_tokens": 20040000, "grad_norm": 1.109375, "lr": 3e-05, "finish_rate": 0.108, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.9, "t_step_s": 72.8, "t_refresh_s": 0.3, "mem_gb": 10.49}
|
| 234 |
+
{"step": 168, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2041427251155178, "tokens": 120000, "cumulative_loss_tokens": 20160000, "grad_norm": 1.1796875, "lr": 3e-05, "finish_rate": 0.117, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 35.2, "t_step_s": 71.1, "t_refresh_s": 0.3, "mem_gb": 10.39}
|
| 235 |
+
{"step": 169, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.2007420184203113, "tokens": 120000, "cumulative_loss_tokens": 20280000, "grad_norm": 1.0546875, "lr": 3e-05, "finish_rate": 0.113, "comp_len": 504.2, "t_data_s": 0.0, "t_rollout_s": 36.2, "t_step_s": 72.4, "t_refresh_s": 0.3, "mem_gb": 10.51}
|
| 236 |
+
{"step": 170, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22800080033155778, "tokens": 120000, "cumulative_loss_tokens": 20400000, "grad_norm": 1.40625, "lr": 3e-05, "finish_rate": 0.153, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 33.6, "t_step_s": 69.9, "t_refresh_s": 0.3, "mem_gb": 10.42}
|
| 237 |
+
[eval step 170] sample: 'To solve this problem, we need to carefully analyze the constraints and use reasoning to determine the maximum number of maples that can be planted.\n\n### Problem Breakdown:\n\n1. **Total Number of Trees'
|
| 238 |
+
{"step": 170, "gsm8k_n": 64, "gsm8k_quick_chat": 0.34375, "t_eval_s": 44.8}
|
| 239 |
+
{"step": 171, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.20762226390354335, "tokens": 120000, "cumulative_loss_tokens": 20520000, "grad_norm": 1.2421875, "lr": 3e-05, "finish_rate": 0.112, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 34.9, "t_step_s": 72.1, "t_refresh_s": 0.3, "mem_gb": 12.43}
|
| 240 |
+
{"step": 172, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.19745559071643898, "tokens": 120000, "cumulative_loss_tokens": 20640000, "grad_norm": 1.1875, "lr": 3e-05, "finish_rate": 0.145, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 33.4, "t_step_s": 69.5, "t_refresh_s": 0.3, "mem_gb": 10.38}
|
healed/healing_breadth/glean_math_keep25_seed1224_long768.console.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/healing_breadth/keep25_long.queue.log
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
2026-07-15T06:33:26-07:00 resuming keep25 at step 50; target step 500 (60000000 cumulative tokens)
|
healed/healing_breadth/reap_math_keep75_seed1224.console.log
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 2 |
+
warnings.warn('Grouped GEMM not available.')
|
| 3 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/henry/.netrc.
|
| 4 |
+
wandb: Currently logged in as: hbfreed to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 5 |
+
wandb: Tracking run with wandb version 0.28.0
|
| 6 |
+
wandb: Run data is saved locally in outputs/healed/healing_breadth/reap_math_keep75_seed1224/wandb/run-20260714_215914-wghixdgk
|
| 7 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 8 |
+
wandb: Syncing run heal-reap_math_keep75-seed1224-run1
|
| 9 |
+
wandb: ⭐️ View project at https://wandb.ai/hbfreed/glean-heal
|
| 10 |
+
wandb: 🚀 View run at https://wandb.ai/hbfreed/glean-heal/runs/wghixdgk
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
starting vllm rollout server on GPU GPU-864c54df-0130-7780-e271-8a5551d1733f (port 8377) ...
|
| 14 |
+
vllm server healthy in 24s
|
| 15 |
+
dataset_source filter ['omega', 'polaris', 'orz_math', 'mathsub', 'dapo-math'] -> 63998 prompts
|
| 16 |
+
|
| 17 |
+
difficulty <= 4 -> 63998 prompts
|
| 18 |
+
WARNING: difficulty filter removed nothing — the slice likely carries difficulty=None (Dolci math sources do), so the teacher-competence guard is NOT in effect
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
63977 prompts | 940 steps/epoch | 50 total steps | student params 5.31B | teacher overlap=True
|
| 22 |
+
{"step": 1, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.22007248604670168, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 4.15625, "lr": 6e-06, "finish_rate": 0.079, "comp_len": 502.1, "t_data_s": 0.0, "t_rollout_s": 31.9, "t_step_s": 123.1, "t_refresh_s": 2.7, "mem_gb": 21.1}
|
| 23 |
+
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
|
| 24 |
+
[eval step 1] sample: 'To solve this problem, we need to maximize the number of maples \\( m \\) while ensuring that no two maples are exactly 5 trees apart.\n\nGiven:\n1. Total number of trees \\( m + l = 75 \\)\n2. No two maples '
|
| 25 |
+
{"step": 2, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.19900615173100183, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 3.890625, "lr": 9e-06, "finish_rate": 0.1, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 32.2, "t_step_s": 105.8, "t_refresh_s": 2.7, "mem_gb": 21.25}
|
| 26 |
+
{"step": 3, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.20272269249390812, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 3.609375, "lr": 1.2e-05, "finish_rate": 0.112, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 32.7, "t_step_s": 106.0, "t_refresh_s": 2.7, "mem_gb": 21.29}
|
| 27 |
+
{"step": 4, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.16433406947404147, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 2.453125, "lr": 1.5e-05, "finish_rate": 0.156, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 31.4, "t_step_s": 102.7, "t_refresh_s": 2.7, "mem_gb": 21.27}
|
| 28 |
+
{"step": 5, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.13169451138923566, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 1.84375, "lr": 1.8e-05, "finish_rate": 0.129, "comp_len": 500.0, "t_data_s": 0.0, "t_rollout_s": 31.5, "t_step_s": 101.5, "t_refresh_s": 2.7, "mem_gb": 21.26}
|
| 29 |
+
{"step": 6, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.11568032757478455, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 2.140625, "lr": 2.1e-05, "finish_rate": 0.176, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 31.3, "t_step_s": 103.0, "t_refresh_s": 2.7, "mem_gb": 21.21}
|
| 30 |
+
{"step": 7, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.09576526515452811, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 1.296875, "lr": 2.4e-05, "finish_rate": 0.23, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 30.8, "t_step_s": 101.4, "t_refresh_s": 2.7, "mem_gb": 21.21}
|
| 31 |
+
{"step": 8, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.08849306978667155, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 1.671875, "lr": 2.7000000000000002e-05, "finish_rate": 0.199, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 31.2, "t_step_s": 103.4, "t_refresh_s": 2.7, "mem_gb": 21.25}
|
| 32 |
+
{"step": 9, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.08143095947749292, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 1.1953125, "lr": 3e-05, "finish_rate": 0.202, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 31.0, "t_step_s": 103.3, "t_refresh_s": 2.7, "mem_gb": 21.21}
|
| 33 |
+
{"step": 10, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.0782286991564557, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 0.94140625, "lr": 3e-05, "finish_rate": 0.143, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 31.6, "t_step_s": 102.8, "t_refresh_s": 2.7, "mem_gb": 21.29}
|
| 34 |
+
[eval step 10] sample: "To solve this problem, we need to maximize the number of maples \\( m \\) such that there are no two maples exactly 5 trees apart. Let's denote the positions of the maples as \\( x_1, x_2, \\ldots, x_m \\)"
|
| 35 |
+
{"step": 11, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.08126618308176597, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 0.9453125, "lr": 3e-05, "finish_rate": 0.196, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 30.9, "t_step_s": 103.0, "t_refresh_s": 2.7, "mem_gb": 21.21}
|
| 36 |
+
{"step": 12, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.07858114459690017, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 0.8828125, "lr": 3e-05, "finish_rate": 0.184, "comp_len": 480.0, "t_data_s": 0.0, "t_rollout_s": 30.9, "t_step_s": 103.6, "t_refresh_s": 2.7, "mem_gb": 21.28}
|
| 37 |
+
{"step": 13, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.06576858285398533, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.7890625, "lr": 3e-05, "finish_rate": 0.164, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 31.4, "t_step_s": 102.8, "t_refresh_s": 2.7, "mem_gb": 21.23}
|
| 38 |
+
{"step": 14, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.07604408309549714, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.96484375, "lr": 3e-05, "finish_rate": 0.213, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 31.2, "t_step_s": 101.5, "t_refresh_s": 2.7, "mem_gb": 21.25}
|
| 39 |
+
{"step": 15, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.0675725906633772, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.8125, "lr": 3e-05, "finish_rate": 0.163, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 31.7, "t_step_s": 103.9, "t_refresh_s": 2.7, "mem_gb": 21.26}
|
| 40 |
+
{"step": 16, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.06715507488114139, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.984375, "lr": 3e-05, "finish_rate": 0.158, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 31.4, "t_step_s": 103.5, "t_refresh_s": 2.7, "mem_gb": 21.24}
|
| 41 |
+
{"step": 17, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.06346533598719786, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 1.046875, "lr": 3e-05, "finish_rate": 0.183, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 31.0, "t_step_s": 101.8, "t_refresh_s": 2.7, "mem_gb": 21.25}
|
| 42 |
+
{"step": 18, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.0625511735424865, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.8203125, "lr": 3e-05, "finish_rate": 0.169, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 31.4, "t_step_s": 102.1, "t_refresh_s": 2.7, "mem_gb": 21.28}
|
| 43 |
+
{"step": 19, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.05953531045786416, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.73046875, "lr": 3e-05, "finish_rate": 0.148, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 31.9, "t_step_s": 103.6, "t_refresh_s": 2.7, "mem_gb": 21.26}
|
| 44 |
+
{"step": 20, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.0648774014983947, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.859375, "lr": 3e-05, "finish_rate": 0.163, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 31.6, "t_step_s": 103.0, "t_refresh_s": 2.7, "mem_gb": 21.27}
|
| 45 |
+
[eval step 20] sample: "To solve this problem, we need to maximize the number of maples \\( m \\) while ensuring that there are no two maples that are exactly 5 trees apart. \n\nLet's break down the problem:\n\n1. **Define Variabl"
|
| 46 |
+
{"step": 21, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.057967641365962726, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 1.0625, "lr": 3e-05, "finish_rate": 0.259, "comp_len": 478.1, "t_data_s": 0.0, "t_rollout_s": 30.2, "t_step_s": 102.4, "t_refresh_s": 2.7, "mem_gb": 21.21}
|
| 47 |
+
{"step": 22, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.058268615744728594, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.6796875, "lr": 3e-05, "finish_rate": 0.186, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 31.3, "t_step_s": 103.7, "t_refresh_s": 2.7, "mem_gb": 21.24}
|
| 48 |
+
{"step": 23, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.052534118128793, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.6484375, "lr": 3e-05, "finish_rate": 0.153, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 31.3, "t_step_s": 101.6, "t_refresh_s": 2.7, "mem_gb": 21.24}
|
| 49 |
+
{"step": 24, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.06487227020380087, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.8203125, "lr": 3e-05, "finish_rate": 0.129, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 31.9, "t_step_s": 103.5, "t_refresh_s": 2.7, "mem_gb": 21.26}
|
| 50 |
+
{"step": 25, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.06092730059704433, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.75, "lr": 3e-05, "finish_rate": 0.184, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 31.9, "t_step_s": 103.7, "t_refresh_s": 2.7, "mem_gb": 21.29}
|
| 51 |
+
checkpoint snapshot queued -> outputs/healed/healing_breadth/reap_math_keep75_seed1224/step0025
|
| 52 |
+
{"step": 26, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.062316528513189404, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 1.015625, "lr": 3e-05, "finish_rate": 0.17, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 31.6, "t_step_s": 104.3, "t_refresh_s": 2.7, "mem_gb": 21.29}
|
| 53 |
+
{"step": 27, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.05730062984156733, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.75, "lr": 3e-05, "finish_rate": 0.14, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 32.3, "t_step_s": 105.0, "t_refresh_s": 2.7, "mem_gb": 21.32}
|
| 54 |
+
{"step": 28, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.05523895585861367, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.734375, "lr": 3e-05, "finish_rate": 0.129, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 31.7, "t_step_s": 102.8, "t_refresh_s": 2.7, "mem_gb": 21.27}
|
| 55 |
+
{"step": 29, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.05680957672783794, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.7265625, "lr": 3e-05, "finish_rate": 0.127, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 32.0, "t_step_s": 104.9, "t_refresh_s": 2.7, "mem_gb": 21.29}
|
| 56 |
+
{"step": 30, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.059886360732543595, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.68359375, "lr": 3e-05, "finish_rate": 0.227, "comp_len": 485.8, "t_data_s": 0.0, "t_rollout_s": 30.8, "t_step_s": 102.0, "t_refresh_s": 2.7, "mem_gb": 21.25}
|
| 57 |
+
[eval step 30] sample: "To solve this problem, we need to determine the maximum number of maples (\\(M\\)) that can be planted along the alley such that there are no two maples exactly 5 trees apart.\n\nLet's break down the prob"
|
| 58 |
+
{"step": 31, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.05055032962113619, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.213, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 31.0, "t_step_s": 102.2, "t_refresh_s": 2.7, "mem_gb": 21.34}
|
| 59 |
+
{"step": 32, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.05529377434644848, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.6796875, "lr": 3e-05, "finish_rate": 0.167, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 31.4, "t_step_s": 103.0, "t_refresh_s": 2.7, "mem_gb": 21.26}
|
| 60 |
+
{"step": 33, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.058952488194188725, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.76171875, "lr": 3e-05, "finish_rate": 0.195, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 31.4, "t_step_s": 103.7, "t_refresh_s": 2.7, "mem_gb": 21.29}
|
| 61 |
+
{"step": 34, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.05145253275079498, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.81640625, "lr": 3e-05, "finish_rate": 0.24, "comp_len": 487.8, "t_data_s": 0.0, "t_rollout_s": 31.1, "t_step_s": 103.0, "t_refresh_s": 2.7, "mem_gb": 21.28}
|
| 62 |
+
{"step": 35, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.05340613210907516, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.7578125, "lr": 3e-05, "finish_rate": 0.253, "comp_len": 474.3, "t_data_s": 0.0, "t_rollout_s": 31.0, "t_step_s": 105.5, "t_refresh_s": 2.7, "mem_gb": 21.27}
|
| 63 |
+
{"step": 36, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.04817226686875026, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.7890625, "lr": 3e-05, "finish_rate": 0.18, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 32.0, "t_step_s": 104.0, "t_refresh_s": 2.7, "mem_gb": 21.27}
|
| 64 |
+
{"step": 37, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.054132809651192895, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.66015625, "lr": 3e-05, "finish_rate": 0.159, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 31.5, "t_step_s": 104.0, "t_refresh_s": 2.7, "mem_gb": 21.24}
|
| 65 |
+
{"step": 38, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.05171422024030859, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 1.1484375, "lr": 3e-05, "finish_rate": 0.178, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 31.4, "t_step_s": 101.7, "t_refresh_s": 2.7, "mem_gb": 21.27}
|
| 66 |
+
{"step": 39, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.05344836845624571, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.16, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 31.3, "t_step_s": 102.8, "t_refresh_s": 2.7, "mem_gb": 21.26}
|
| 67 |
+
{"step": 40, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.050235971879477925, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.62890625, "lr": 3e-05, "finish_rate": 0.165, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 31.9, "t_step_s": 103.1, "t_refresh_s": 2.7, "mem_gb": 21.26}
|
| 68 |
+
[eval step 40] sample: "To solve this problem, we need to maximize the number of maples \\( m \\) such that there are no two maples exactly 5 trees apart. Let's denote the number of larches by \\( l \\). According to the problem"
|
| 69 |
+
{"step": 41, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.04683446280825883, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.79296875, "lr": 3e-05, "finish_rate": 0.181, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 31.3, "t_step_s": 104.6, "t_refresh_s": 2.7, "mem_gb": 21.33}
|
| 70 |
+
{"step": 42, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.051344527876703065, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.147, "comp_len": 489.8, "t_data_s": 0.0, "t_rollout_s": 31.5, "t_step_s": 103.0, "t_refresh_s": 2.7, "mem_gb": 21.25}
|
| 71 |
+
{"step": 43, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.0487479245382982, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.84375, "lr": 3e-05, "finish_rate": 0.172, "comp_len": 491.8, "t_data_s": 0.0, "t_rollout_s": 31.1, "t_step_s": 101.6, "t_refresh_s": 2.7, "mem_gb": 21.22}
|
| 72 |
+
{"step": 44, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.04652415793016553, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.69140625, "lr": 3e-05, "finish_rate": 0.197, "comp_len": 481.9, "t_data_s": 0.0, "t_rollout_s": 30.9, "t_step_s": 102.8, "t_refresh_s": 2.7, "mem_gb": 21.28}
|
| 73 |
+
{"step": 45, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.052038926035289965, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.87109375, "lr": 3e-05, "finish_rate": 0.152, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 31.1, "t_step_s": 101.6, "t_refresh_s": 2.7, "mem_gb": 21.2}
|
| 74 |
+
{"step": 46, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.051263596718541034, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.65234375, "lr": 3e-05, "finish_rate": 0.173, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 31.5, "t_step_s": 103.1, "t_refresh_s": 2.7, "mem_gb": 21.24}
|
| 75 |
+
{"step": 47, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.051921346824189335, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.74609375, "lr": 3e-05, "finish_rate": 0.219, "comp_len": 478.1, "t_data_s": 0.0, "t_rollout_s": 30.8, "t_step_s": 103.4, "t_refresh_s": 2.7, "mem_gb": 21.25}
|
| 76 |
+
{"step": 48, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.05030671559885765, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.65625, "lr": 3e-05, "finish_rate": 0.136, "comp_len": 495.9, "t_data_s": 0.0, "t_rollout_s": 31.9, "t_step_s": 103.4, "t_refresh_s": 2.7, "mem_gb": 21.24}
|
| 77 |
+
{"step": 49, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.048079004754545164, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.59375, "lr": 3e-05, "finish_rate": 0.095, "comp_len": 497.9, "t_data_s": 0.0, "t_rollout_s": 31.6, "t_step_s": 102.4, "t_refresh_s": 2.7, "mem_gb": 21.28}
|
| 78 |
+
{"step": 50, "epoch": 0, "training_mode": "on-policy", "reverse_kl": 0.051015459259475274, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.78515625, "lr": 3e-05, "finish_rate": 0.136, "comp_len": 493.8, "t_data_s": 0.0, "t_rollout_s": 32.2, "t_step_s": 84.7, "t_refresh_s": 0.0, "mem_gb": 21.29}
|
| 79 |
+
[eval step 50] sample: "To solve this problem, we need to maximize the number of maples \\( m \\) such that there are no two maples exactly 5 trees apart. Let's break down the problem step-by-step:\n\n1. **Understand the Constra"
|
| 80 |
+
checkpoint snapshot queued -> outputs/healed/healing_breadth/reap_math_keep75_seed1224/step0050
|
| 81 |
+
wandb: updating run metadata
|
| 82 |
+
wandb: uploading config.yaml; uploading output.log; uploading wandb-summary.json
|
| 83 |
+
wandb:
|
| 84 |
+
wandb: Run history:
|
| 85 |
+
wandb: comp_len █▇▆█▅▄▃▅▂▂▅▅▃▄▆▅▁▃▇▇▃▇▇▅▃▄▄▄▅▅▅▆▂▅▅▆▆▁▇▆
|
| 86 |
+
wandb: cumulative_loss_tokens ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▅▆▆▆▆▆▇▇▇▇▇▇███
|
| 87 |
+
wandb: epoch ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 88 |
+
wandb: finish_rate ▁▂▂▃▅▆▆▃▆▅▄▄▅▄▄█▅▄▃▅▃▃▃▇▆▆▇█▅▄▄▄▅▄▅▄▅▆▃▃
|
| 89 |
+
wandb: grad_norm █▇▇▅▃▂▃▂▂▂▁▂▁▂▂▁▁▂▁▁▁▂▁▁▁▁▁▁▁▁▂▁▁▁▁▁▁▁▁▁
|
| 90 |
+
wandb: lr ▁▂▃▄▅▆▇█████████████████████████████████
|
| 91 |
+
wandb: mem_gb ▁▅▇▆▆▄▅▄▄▆▅▆▅▅▆▆▄▅▅▆▇▇▆▇█▇▆▆▆▅▆▆█▅▄▄▅▅▅▇
|
| 92 |
+
wandb: reverse_kl █▇▇▆▄▃▃▂▂▂▂▂▂▂▂▂▁▁▁▂▂▁▁▁▂▁▂▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 93 |
+
wandb: step ▁▁▁▁▂▂▂▂▂▂▃▃▃▃▃▄▄▄▄▄▅▅▅▅▅▆▆▆▆▆▆▇▇▇▇▇▇███
|
| 94 |
+
wandb: t_data_s ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁
|
| 95 |
+
wandb: +4 ...
|
| 96 |
+
wandb:
|
| 97 |
+
wandb: Run summary:
|
| 98 |
+
wandb: comp_len 493.8
|
| 99 |
+
wandb: cumulative_loss_tokens 6000000
|
| 100 |
+
wandb: epoch 0
|
| 101 |
+
wandb: finish_rate 0.136
|
| 102 |
+
wandb: grad_norm 0.78516
|
| 103 |
+
wandb: lr 3e-05
|
| 104 |
+
wandb: mem_gb 21.29
|
| 105 |
+
wandb: reverse_kl 0.05102
|
| 106 |
+
wandb: step 50
|
| 107 |
+
wandb: t_data_s 0
|
| 108 |
+
wandb: +5 ...
|
| 109 |
+
wandb:
|
| 110 |
+
wandb: 🚀 View run heal-reap_math_keep75-seed1224-run1 at: https://wandb.ai/hbfreed/glean-heal/runs/wghixdgk
|
| 111 |
+
wandb: ⭐️ View project at: https://wandb.ai/hbfreed/glean-heal
|
| 112 |
+
wandb: Synced 5 W&B file(s), 0 media file(s), 0 artifact file(s) and 0 other file(s)
|
| 113 |
+
wandb: Find logs at: outputs/healed/healing_breadth/reap_math_keep75_seed1224/wandb/run-20260714_215914-wghixdgk/logs
|
healed/healing_breadth/traj_sweep.log
ADDED
|
@@ -0,0 +1,88 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-15T19:40:43-07:00 eval step0025 chat
|
| 2 |
+
{
|
| 3 |
+
"correct": 407,
|
| 4 |
+
"accuracy": 0.30856709628506446,
|
| 5 |
+
"finished": 1008,
|
| 6 |
+
"finish_rate": 0.7642153146322972,
|
| 7 |
+
"mean_completion_tokens": 351.8074298711145
|
| 8 |
+
}
|
| 9 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0025_chat.json
|
| 10 |
+
2026-07-15T19:44:18-07:00 eval step0050 chat
|
| 11 |
+
{
|
| 12 |
+
"correct": 344,
|
| 13 |
+
"accuracy": 0.2608036391205459,
|
| 14 |
+
"finished": 755,
|
| 15 |
+
"finish_rate": 0.5724033358605004,
|
| 16 |
+
"mean_completion_tokens": 369.57088703563306
|
| 17 |
+
}
|
| 18 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0050_chat.json
|
| 19 |
+
2026-07-15T19:47:57-07:00 eval step0100 chat
|
| 20 |
+
{
|
| 21 |
+
"correct": 564,
|
| 22 |
+
"accuracy": 0.4275966641394996,
|
| 23 |
+
"finished": 1235,
|
| 24 |
+
"finish_rate": 0.9363153904473086,
|
| 25 |
+
"mean_completion_tokens": 271.9120545868082
|
| 26 |
+
}
|
| 27 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0100_chat.json
|
| 28 |
+
2026-07-15T19:50:53-07:00 eval step0150 chat
|
| 29 |
+
{
|
| 30 |
+
"correct": 607,
|
| 31 |
+
"accuracy": 0.46019711902956784,
|
| 32 |
+
"finished": 1250,
|
| 33 |
+
"finish_rate": 0.9476876421531463,
|
| 34 |
+
"mean_completion_tokens": 256.28127369219106
|
| 35 |
+
}
|
| 36 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0150_chat.json
|
| 37 |
+
2026-07-15T19:53:43-07:00 eval step0200 chat
|
| 38 |
+
{
|
| 39 |
+
"correct": 606,
|
| 40 |
+
"accuracy": 0.45943896891584535,
|
| 41 |
+
"finished": 1251,
|
| 42 |
+
"finish_rate": 0.9484457922668689,
|
| 43 |
+
"mean_completion_tokens": 245.95754359363153
|
| 44 |
+
}
|
| 45 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0200_chat.json
|
| 46 |
+
2026-07-15T19:56:25-07:00 eval step0250 chat
|
| 47 |
+
{
|
| 48 |
+
"correct": 610,
|
| 49 |
+
"accuracy": 0.4624715693707354,
|
| 50 |
+
"finished": 1225,
|
| 51 |
+
"finish_rate": 0.9287338893100834,
|
| 52 |
+
"mean_completion_tokens": 256.76648976497347
|
| 53 |
+
}
|
| 54 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0250_chat.json
|
| 55 |
+
2026-07-15T19:59:11-07:00 eval step0300 chat
|
| 56 |
+
{
|
| 57 |
+
"correct": 625,
|
| 58 |
+
"accuracy": 0.47384382107657314,
|
| 59 |
+
"finished": 1240,
|
| 60 |
+
"finish_rate": 0.9401061410159212,
|
| 61 |
+
"mean_completion_tokens": 243.75208491281273
|
| 62 |
+
}
|
| 63 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0300_chat.json
|
| 64 |
+
2026-07-15T20:01:52-07:00 step0350 exists, skip
|
| 65 |
+
2026-07-15T20:01:52-07:00 step0400 exists, skip
|
| 66 |
+
2026-07-15T20:01:52-07:00 step0450 exists, skip
|
| 67 |
+
2026-07-15T20:01:52-07:00 eval step0500 chat
|
| 68 |
+
{
|
| 69 |
+
"correct": 631,
|
| 70 |
+
"accuracy": 0.4783927217589083,
|
| 71 |
+
"finished": 1231,
|
| 72 |
+
"finish_rate": 0.9332827899924185,
|
| 73 |
+
"mean_completion_tokens": 265.1554207733131
|
| 74 |
+
}
|
| 75 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0500_chat.json
|
| 76 |
+
2026-07-15T20:04:42-07:00 === keep25 chat healing trajectory (full 1319) ===
|
| 77 |
+
2026-07-15T20:04:42-07:00 step 0025: acc=0.30856709628506446 finish=0.7642153146322972
|
| 78 |
+
2026-07-15T20:04:42-07:00 step 0050: acc=0.2608036391205459 finish=0.5724033358605004
|
| 79 |
+
2026-07-15T20:04:42-07:00 step 0100: acc=0.4275966641394996 finish=0.9363153904473086
|
| 80 |
+
2026-07-15T20:04:42-07:00 step 0150: acc=0.46019711902956784 finish=0.9476876421531463
|
| 81 |
+
2026-07-15T20:04:43-07:00 step 0200: acc=0.45943896891584535 finish=0.9484457922668689
|
| 82 |
+
2026-07-15T20:04:43-07:00 step 0250: acc=0.4624715693707354 finish=0.9287338893100834
|
| 83 |
+
2026-07-15T20:04:43-07:00 step 0300: acc=0.47384382107657314 finish=0.9401061410159212
|
| 84 |
+
2026-07-15T20:04:43-07:00 step 0350: acc=0.48218347232752085 finish=0.956027293404094
|
| 85 |
+
2026-07-15T20:04:43-07:00 step 0400: acc=0.4981046247156937 finish=0.9514783927217589
|
| 86 |
+
2026-07-15T20:04:43-07:00 step 0450: acc=0.48597422289613346 finish=0.9446550416982562
|
| 87 |
+
2026-07-15T20:04:43-07:00 step 0500: acc=0.4783927217589083 finish=0.9332827899924185
|
| 88 |
+
2026-07-15T20:04:43-07:00 TRAJECTORY SWEEP COMPLETE
|
healed/healing_breadth/traj_sweep2.log
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-15T19:51:42-07:00 [w2] eval step0350 chat (gpu 8ca7, port 8378)
|
| 2 |
+
{
|
| 3 |
+
"correct": 636,
|
| 4 |
+
"accuracy": 0.48218347232752085,
|
| 5 |
+
"finished": 1261,
|
| 6 |
+
"finish_rate": 0.956027293404094,
|
| 7 |
+
"mean_completion_tokens": 249.04094010614102
|
| 8 |
+
}
|
| 9 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0350_chat.json
|
| 10 |
+
2026-07-15T19:54:25-07:00 [w2] eval step0400 chat (gpu 8ca7, port 8378)
|
| 11 |
+
{
|
| 12 |
+
"correct": 657,
|
| 13 |
+
"accuracy": 0.4981046247156937,
|
| 14 |
+
"finished": 1255,
|
| 15 |
+
"finish_rate": 0.9514783927217589,
|
| 16 |
+
"mean_completion_tokens": 249.73692191053829
|
| 17 |
+
}
|
| 18 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0400_chat.json
|
| 19 |
+
2026-07-15T19:57:08-07:00 [w2] eval step0450 chat (gpu 8ca7, port 8378)
|
| 20 |
+
{
|
| 21 |
+
"correct": 641,
|
| 22 |
+
"accuracy": 0.48597422289613346,
|
| 23 |
+
"finished": 1246,
|
| 24 |
+
"finish_rate": 0.9446550416982562,
|
| 25 |
+
"mean_completion_tokens": 256.36618650492795
|
| 26 |
+
}
|
| 27 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0450_chat.json
|
| 28 |
+
2026-07-15T19:59:54-07:00 [w2] eval step0500 chat (gpu 8ca7, port 8378)
|
| 29 |
+
{
|
| 30 |
+
"correct": 627,
|
| 31 |
+
"accuracy": 0.47536012130401817,
|
| 32 |
+
"finished": 1239,
|
| 33 |
+
"finish_rate": 0.9393479909021987,
|
| 34 |
+
"mean_completion_tokens": 264.5200909780136
|
| 35 |
+
}
|
| 36 |
+
saved item-level results -> outputs/evals/healing_breadth/keep25_traj/step0500_chat.json
|
| 37 |
+
2026-07-15T20:02:42-07:00 [w2] worker2 tail complete
|
healed/keep50_warmup_fixed_s1224/args.json
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"student": "outputs/pruned/glean-0125inst-math-keep50",
|
| 3 |
+
"teacher": "allenai/OLMoE-1B-7B-0125-Instruct",
|
| 4 |
+
"training_mode": "off-policy",
|
| 5 |
+
"kl_direction": "forward",
|
| 6 |
+
"dataset": "allenai/RLVR-MATH",
|
| 7 |
+
"dataset_sources": null,
|
| 8 |
+
"max_difficulty": null,
|
| 9 |
+
"trajectories": "outputs/teacher_trajectories/dolci_math_curated.jsonl",
|
| 10 |
+
"trajectory_dataset": "allenai/Dolci-Instruct-RL",
|
| 11 |
+
"off_policy_frames": "chat",
|
| 12 |
+
"off_policy_max_seq_len": 2048,
|
| 13 |
+
"topk_targets": "outputs/teacher_trajectories/dolci_math_curated_opd_top128",
|
| 14 |
+
"max_loss_tokens": null,
|
| 15 |
+
"loss_tokens_per_step": 120000,
|
| 16 |
+
"teacher_device": "cuda:0",
|
| 17 |
+
"student_device": "cuda:0",
|
| 18 |
+
"lr": 3e-05,
|
| 19 |
+
"optimizer": "adamw8bit",
|
| 20 |
+
"weight_decay": 0.1,
|
| 21 |
+
"epochs": 3,
|
| 22 |
+
"prompts_per_step": 256,
|
| 23 |
+
"group_size": 1,
|
| 24 |
+
"rollout_batch": 64,
|
| 25 |
+
"micro_batch": 3,
|
| 26 |
+
"max_new_tokens": 256,
|
| 27 |
+
"max_prompt_len": 1024,
|
| 28 |
+
"warmup_steps": 10,
|
| 29 |
+
"max_grad_norm": 1.0,
|
| 30 |
+
"eval_every": 10,
|
| 31 |
+
"gsm8k_every": 0,
|
| 32 |
+
"gsm8k_n": 256,
|
| 33 |
+
"gsm8k_batch": 16,
|
| 34 |
+
"gsm8k_max_new_tokens": 512,
|
| 35 |
+
"gsm8k_frames": "chat",
|
| 36 |
+
"save_every": 50,
|
| 37 |
+
"out_dir": "outputs/healed/keep50_warmup_fixed_s1224",
|
| 38 |
+
"sweep": 150,
|
| 39 |
+
"wandb": true,
|
| 40 |
+
"wandb_project": "glean-heal",
|
| 41 |
+
"wandb_run_name": "warmup-fixed-keep50-s1224",
|
| 42 |
+
"wandb_run_id": null,
|
| 43 |
+
"wandb_resume": null,
|
| 44 |
+
"wandb_mode": "online",
|
| 45 |
+
"no_wandb_sync": false,
|
| 46 |
+
"debug": false,
|
| 47 |
+
"resume_from": null,
|
| 48 |
+
"start_step": 0,
|
| 49 |
+
"no_grad_checkpointing": false,
|
| 50 |
+
"seed": 1224,
|
| 51 |
+
"no_teacher_overlap": false,
|
| 52 |
+
"sync_checkpoints": false,
|
| 53 |
+
"rollout_engine": "hf",
|
| 54 |
+
"vllm_gpu": null,
|
| 55 |
+
"vllm_port": 8377,
|
| 56 |
+
"vllm_refresh_every": 5,
|
| 57 |
+
"vllm_serve_bin": "vllm-plugin/.venv/bin/python",
|
| 58 |
+
"vllm_gpu_mem_util": 0.85,
|
| 59 |
+
"liger_loss": false,
|
| 60 |
+
"gold_mix_lambda": 0.0,
|
| 61 |
+
"gold_topk_targets": null,
|
| 62 |
+
"gold_loss": "ce",
|
| 63 |
+
"gold_mix_decay": 0.0,
|
| 64 |
+
"fast_teacher": false,
|
| 65 |
+
"fast_student": false,
|
| 66 |
+
"reference_kl_beta": 0.0,
|
| 67 |
+
"drop_truncated_rollouts": false,
|
| 68 |
+
"vllm_max_model_len": null,
|
| 69 |
+
"vllm_refresh_mode": "reload",
|
| 70 |
+
"vllm_live_dir": null,
|
| 71 |
+
"resolved_kl_direction": "forward"
|
| 72 |
+
}
|
healed/keep50_warmup_fixed_s1224/step0150/modeling_pruned_olmoe.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""GLEAN-pruned OLMoE: HF-loadable model with ragged (variable-width) experts.
|
| 2 |
+
|
| 3 |
+
Pattern follows hbfreed/variable-flex-olmo's PrunedFlexOlmoForCausalLM
|
| 4 |
+
(docs/recon/prior-work-hbfreed.md), generalized from one scalar width to a
|
| 5 |
+
per-(layer, expert) width table: ``super().__init__`` builds the uniform
|
| 6 |
+
architecture from the config, then every MoE block is rebuilt to its pruned
|
| 7 |
+
shape — surviving experts only, each at its own width, router sliced to
|
| 8 |
+
match — so the state dict aligns exactly with what
|
| 9 |
+
``glean.prune.prune_channels_global`` leaves behind.
|
| 10 |
+
|
| 11 |
+
Caveat: ``output_router_logits=True`` (the load-balancing aux loss) assumes a
|
| 12 |
+
uniform ``config.num_experts`` and is unsupported on ragged models.
|
| 13 |
+
"""
|
| 14 |
+
|
| 15 |
+
import torch.nn as nn
|
| 16 |
+
from transformers.activations import ACT2FN
|
| 17 |
+
from transformers.models.olmoe.modeling_olmoe import OlmoeForCausalLM
|
| 18 |
+
|
| 19 |
+
from .configuration_pruned_olmoe import PrunedOlmoeConfig
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
class RaggedOlmoeMLP(nn.Module):
|
| 23 |
+
"""OlmoeMLP with an explicit intermediate width (SwiGLU, no biases)."""
|
| 24 |
+
|
| 25 |
+
def __init__(self, hidden_size: int, intermediate_size: int, hidden_act: str):
|
| 26 |
+
super().__init__()
|
| 27 |
+
self.hidden_size = hidden_size
|
| 28 |
+
self.intermediate_size = intermediate_size
|
| 29 |
+
self.gate_proj = nn.Linear(hidden_size, intermediate_size, bias=False)
|
| 30 |
+
self.up_proj = nn.Linear(hidden_size, intermediate_size, bias=False)
|
| 31 |
+
self.down_proj = nn.Linear(intermediate_size, hidden_size, bias=False)
|
| 32 |
+
self.act_fn = ACT2FN[hidden_act]
|
| 33 |
+
|
| 34 |
+
def forward(self, x):
|
| 35 |
+
return self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 36 |
+
|
| 37 |
+
|
| 38 |
+
class PrunedOlmoeForCausalLM(OlmoeForCausalLM):
|
| 39 |
+
"""OLMoE with per-layer surviving-expert lists at per-expert widths."""
|
| 40 |
+
|
| 41 |
+
config_class = PrunedOlmoeConfig
|
| 42 |
+
|
| 43 |
+
def __init__(self, config: PrunedOlmoeConfig):
|
| 44 |
+
super().__init__(config)
|
| 45 |
+
widths_table = getattr(config, "expert_widths", None)
|
| 46 |
+
if widths_table is None:
|
| 47 |
+
return # unpruned: plain OLMoE
|
| 48 |
+
if len(widths_table) != len(self.model.layers):
|
| 49 |
+
raise ValueError(
|
| 50 |
+
f"expert_widths has {len(widths_table)} rows but the model has "
|
| 51 |
+
f"{len(self.model.layers)} decoder layers"
|
| 52 |
+
)
|
| 53 |
+
for layer, widths in zip(self.model.layers, widths_table):
|
| 54 |
+
if any(w <= 0 for w in widths):
|
| 55 |
+
raise ValueError("expert_widths must list surviving experts only (>0)")
|
| 56 |
+
block = layer.mlp
|
| 57 |
+
if len(widths) < block.top_k:
|
| 58 |
+
raise ValueError(
|
| 59 |
+
f"a layer keeps {len(widths)} experts < top_k={block.top_k}"
|
| 60 |
+
)
|
| 61 |
+
block.num_experts = len(widths)
|
| 62 |
+
block.gate = nn.Linear(config.hidden_size, len(widths), bias=False)
|
| 63 |
+
block.experts = nn.ModuleList(
|
| 64 |
+
RaggedOlmoeMLP(config.hidden_size, w, config.hidden_act)
|
| 65 |
+
for w in widths
|
| 66 |
+
)
|
healed/keep50_warmup_fixed_s1224/train_log.jsonl
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"step": 1, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21896971870896717, "tokens": 120000, "cumulative_loss_tokens": 120000, "grad_norm": 4.6875, "lr": 6e-06, "finish_rate": 0.907, "comp_len": 508.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.337, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 99.1, "frames": {"chat": 236}, "mem_gb": 15.78, "mem_gb_teacher": 15.78}
|
| 2 |
+
{"step": 2, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.27215903437460465, "tokens": 120000, "cumulative_loss_tokens": 240000, "grad_norm": 4.84375, "lr": 9e-06, "finish_rate": 0.781, "comp_len": 558.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.474, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.6, "frames": {"chat": 215}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 3 |
+
{"step": 3, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2780314308715363, "tokens": 120000, "cumulative_loss_tokens": 360000, "grad_norm": 4.09375, "lr": 1.2e-05, "finish_rate": 0.825, "comp_len": 553.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.376, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.6, "frames": {"chat": 217}, "mem_gb": 15.93, "mem_gb_teacher": 15.93}
|
| 4 |
+
{"step": 4, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.21216022065331538, "tokens": 120000, "cumulative_loss_tokens": 480000, "grad_norm": 2.546875, "lr": 1.5e-05, "finish_rate": 0.8, "comp_len": 585.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.366, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 86.0, "frames": {"chat": 205}, "mem_gb": 15.99, "mem_gb_teacher": 15.99}
|
| 5 |
+
{"step": 5, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.13948054732351253, "tokens": 120000, "cumulative_loss_tokens": 600000, "grad_norm": 1.6484375, "lr": 1.8e-05, "finish_rate": 0.834, "comp_len": 524.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.318, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 94.5, "frames": {"chat": 229}, "mem_gb": 15.96, "mem_gb_teacher": 15.96}
|
| 6 |
+
{"step": 6, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.2525411546646928, "tokens": 120000, "cumulative_loss_tokens": 720000, "grad_norm": 2.578125, "lr": 2.1e-05, "finish_rate": 0.812, "comp_len": 538.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.336, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 92.0, "frames": {"chat": 223}, "mem_gb": 16.03, "mem_gb_teacher": 16.03}
|
| 7 |
+
{"step": 7, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1408244575532774, "tokens": 120000, "cumulative_loss_tokens": 840000, "grad_norm": 1.171875, "lr": 2.4e-05, "finish_rate": 0.708, "comp_len": 594.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.42, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 86.3, "frames": {"chat": 202}, "mem_gb": 16.07, "mem_gb_teacher": 16.07}
|
| 8 |
+
{"step": 8, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.16170320059585697, "tokens": 120000, "cumulative_loss_tokens": 960000, "grad_norm": 1.171875, "lr": 2.7000000000000002e-05, "finish_rate": 0.77, "comp_len": 574.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.453, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 87.6, "frames": {"chat": 209}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 9 |
+
{"step": 9, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1060452919805112, "tokens": 120000, "cumulative_loss_tokens": 1080000, "grad_norm": 0.703125, "lr": 3e-05, "finish_rate": 0.885, "comp_len": 528.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.397, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 93.7, "frames": {"chat": 227}, "mem_gb": 16.02, "mem_gb_teacher": 16.02}
|
| 10 |
+
{"step": 10, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.11166963671234746, "tokens": 120000, "cumulative_loss_tokens": 1200000, "grad_norm": 0.66796875, "lr": 3e-05, "finish_rate": 0.848, "comp_len": 521.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.546, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 95.5, "frames": {"chat": 230}, "mem_gb": 16.09, "mem_gb_teacher": 16.09}
|
| 11 |
+
{"step": 11, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.09883926218959192, "tokens": 120000, "cumulative_loss_tokens": 1320000, "grad_norm": 0.671875, "lr": 3e-05, "finish_rate": 0.879, "comp_len": 519.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.426, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 94.7, "frames": {"chat": 231}, "mem_gb": 15.94, "mem_gb_teacher": 15.94}
|
| 12 |
+
{"step": 12, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.09610702018613616, "tokens": 120000, "cumulative_loss_tokens": 1440000, "grad_norm": 0.55078125, "lr": 3e-05, "finish_rate": 0.882, "comp_len": 489.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.246, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 99.3, "frames": {"chat": 245}, "mem_gb": 16.02, "mem_gb_teacher": 16.02}
|
| 13 |
+
{"step": 13, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.09665392311389248, "tokens": 120000, "cumulative_loss_tokens": 1560000, "grad_norm": 0.5703125, "lr": 3e-05, "finish_rate": 0.81, "comp_len": 571.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.375, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 87.0, "frames": {"chat": 210}, "mem_gb": 16.02, "mem_gb_teacher": 16.02}
|
| 14 |
+
{"step": 14, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.1129512736000431, "tokens": 120000, "cumulative_loss_tokens": 1680000, "grad_norm": 0.578125, "lr": 3e-05, "finish_rate": 0.758, "comp_len": 568.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.312, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.4, "frames": {"chat": 211}, "mem_gb": 16.02, "mem_gb_teacher": 16.02}
|
| 15 |
+
{"step": 15, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.09338119786353782, "tokens": 120000, "cumulative_loss_tokens": 1800000, "grad_norm": 0.453125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.413, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 92.0, "frames": {"chat": 221}, "mem_gb": 16.08, "mem_gb_teacher": 16.08}
|
| 16 |
+
{"step": 16, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08123978671409811, "tokens": 120000, "cumulative_loss_tokens": 1920000, "grad_norm": 0.490234375, "lr": 3e-05, "finish_rate": 0.912, "comp_len": 480.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.366, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 101.9, "frames": {"chat": 250}, "mem_gb": 15.89, "mem_gb_teacher": 15.89}
|
| 17 |
+
{"step": 17, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08966803371421993, "tokens": 120000, "cumulative_loss_tokens": 2040000, "grad_norm": 0.49609375, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 524.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.402, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 95.6, "frames": {"chat": 229}, "mem_gb": 16.06, "mem_gb_teacher": 16.06}
|
| 18 |
+
{"step": 18, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06860631760672356, "tokens": 120000, "cumulative_loss_tokens": 2160000, "grad_norm": 0.408203125, "lr": 3e-05, "finish_rate": 0.888, "comp_len": 480.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.479, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 101.4, "frames": {"chat": 250}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 19 |
+
{"step": 19, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08370754941549773, "tokens": 120000, "cumulative_loss_tokens": 2280000, "grad_norm": 0.421875, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 519.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.363, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 95.2, "frames": {"chat": 231}, "mem_gb": 15.91, "mem_gb_teacher": 15.91}
|
| 20 |
+
{"step": 20, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07473204600410537, "tokens": 120000, "cumulative_loss_tokens": 2400000, "grad_norm": 0.41796875, "lr": 3e-05, "finish_rate": 0.844, "comp_len": 535.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.433, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 92.5, "frames": {"chat": 224}, "mem_gb": 15.96, "mem_gb_teacher": 15.96}
|
| 21 |
+
{"step": 21, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08024157820896556, "tokens": 120000, "cumulative_loss_tokens": 2520000, "grad_norm": 0.486328125, "lr": 3e-05, "finish_rate": 0.802, "comp_len": 566.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.329, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 88.4, "frames": {"chat": 212}, "mem_gb": 16.0, "mem_gb_teacher": 16.0}
|
| 22 |
+
{"step": 22, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0721184289892825, "tokens": 120000, "cumulative_loss_tokens": 2640000, "grad_norm": 0.435546875, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 504.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.376, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 97.3, "frames": {"chat": 238}, "mem_gb": 15.95, "mem_gb_teacher": 15.95}
|
| 23 |
+
{"step": 23, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07248173923180438, "tokens": 120000, "cumulative_loss_tokens": 2760000, "grad_norm": 0.443359375, "lr": 3e-05, "finish_rate": 0.903, "comp_len": 466.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.445, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 103.5, "frames": {"chat": 257}, "mem_gb": 15.83, "mem_gb_teacher": 15.83}
|
| 24 |
+
{"step": 24, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0655906916304181, "tokens": 120000, "cumulative_loss_tokens": 2880000, "grad_norm": 0.36328125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 528.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.405, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 93.3, "frames": {"chat": 227}, "mem_gb": 16.02, "mem_gb_teacher": 16.02}
|
| 25 |
+
{"step": 25, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07533252080177578, "tokens": 120000, "cumulative_loss_tokens": 3000000, "grad_norm": 0.380859375, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 526.3, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.359, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 93.6, "frames": {"chat": 228}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 26 |
+
{"step": 26, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07692283792655605, "tokens": 120000, "cumulative_loss_tokens": 3120000, "grad_norm": 0.40625, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 515.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.542, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 97.0, "frames": {"chat": 233}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 27 |
+
{"step": 27, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06803224476923546, "tokens": 120000, "cumulative_loss_tokens": 3240000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 515.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.351, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 96.2, "frames": {"chat": 233}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 28 |
+
{"step": 28, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.09600000411309302, "tokens": 120000, "cumulative_loss_tokens": 3360000, "grad_norm": 0.46484375, "lr": 3e-05, "finish_rate": 0.731, "comp_len": 609.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.388, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 85.0, "frames": {"chat": 197}, "mem_gb": 16.13, "mem_gb_teacher": 16.13}
|
| 29 |
+
{"step": 29, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0838949200007754, "tokens": 120000, "cumulative_loss_tokens": 3480000, "grad_norm": 0.404296875, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 502.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.384, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 98.3, "frames": {"chat": 239}, "mem_gb": 15.88, "mem_gb_teacher": 15.88}
|
| 30 |
+
{"step": 30, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.08550032369385784, "tokens": 120000, "cumulative_loss_tokens": 3600000, "grad_norm": 0.640625, "lr": 3e-05, "finish_rate": 0.83, "comp_len": 535.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.302, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 91.8, "frames": {"chat": 224}, "mem_gb": 15.93, "mem_gb_teacher": 15.93}
|
| 31 |
+
{"step": 31, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06893708595448794, "tokens": 120000, "cumulative_loss_tokens": 3720000, "grad_norm": 0.3828125, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 553.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.388, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.4, "frames": {"chat": 217}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 32 |
+
{"step": 32, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0662639382578743, "tokens": 120000, "cumulative_loss_tokens": 3840000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.259, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 98.4, "frames": {"chat": 241}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 33 |
+
{"step": 33, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06439438750580885, "tokens": 120000, "cumulative_loss_tokens": 3960000, "grad_norm": 0.39453125, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 550.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.407, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.8, "frames": {"chat": 218}, "mem_gb": 16.02, "mem_gb_teacher": 16.02}
|
| 34 |
+
{"step": 34, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06970151182319968, "tokens": 120000, "cumulative_loss_tokens": 4080000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.439, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 86.2, "frames": {"chat": 206}, "mem_gb": 16.03, "mem_gb_teacher": 16.03}
|
| 35 |
+
{"step": 35, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06962556957538861, "tokens": 120000, "cumulative_loss_tokens": 4200000, "grad_norm": 0.37890625, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 517.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.512, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 96.4, "frames": {"chat": 232}, "mem_gb": 16.07, "mem_gb_teacher": 16.07}
|
| 36 |
+
{"step": 36, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07808773103132843, "tokens": 120000, "cumulative_loss_tokens": 4320000, "grad_norm": 0.419921875, "lr": 3e-05, "finish_rate": 0.771, "comp_len": 550.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.432, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 92.0, "frames": {"chat": 218}, "mem_gb": 16.09, "mem_gb_teacher": 16.09}
|
| 37 |
+
{"step": 37, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07417992857682208, "tokens": 120000, "cumulative_loss_tokens": 4440000, "grad_norm": 0.376953125, "lr": 3e-05, "finish_rate": 0.779, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.428, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 88.9, "frames": {"chat": 213}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 38 |
+
{"step": 38, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07201246263841167, "tokens": 120000, "cumulative_loss_tokens": 4560000, "grad_norm": 0.369140625, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 483.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.527, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 100.9, "frames": {"chat": 248}, "mem_gb": 16.02, "mem_gb_teacher": 16.02}
|
| 39 |
+
{"step": 39, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05988815330729509, "tokens": 120000, "cumulative_loss_tokens": 4680000, "grad_norm": 0.373046875, "lr": 3e-05, "finish_rate": 0.803, "comp_len": 550.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.333, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.1, "frames": {"chat": 218}, "mem_gb": 16.08, "mem_gb_teacher": 16.08}
|
| 40 |
+
{"step": 40, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05963528015368308, "tokens": 120000, "cumulative_loss_tokens": 4800000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.851, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.401, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 91.5, "frames": {"chat": 221}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 41 |
+
{"step": 41, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05816531496203194, "tokens": 120000, "cumulative_loss_tokens": 4920000, "grad_norm": 0.345703125, "lr": 3e-05, "finish_rate": 0.894, "comp_len": 508.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.396, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 96.1, "frames": {"chat": 236}, "mem_gb": 15.97, "mem_gb_teacher": 15.97}
|
| 42 |
+
{"step": 42, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06005277933338657, "tokens": 120000, "cumulative_loss_tokens": 5040000, "grad_norm": 0.326171875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 487.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.338, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 100.6, "frames": {"chat": 246}, "mem_gb": 15.89, "mem_gb_teacher": 15.89}
|
| 43 |
+
{"step": 43, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06171362210111692, "tokens": 120000, "cumulative_loss_tokens": 5160000, "grad_norm": 0.35546875, "lr": 3e-05, "finish_rate": 0.838, "comp_len": 512.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.48, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 96.2, "frames": {"chat": 234}, "mem_gb": 16.14, "mem_gb_teacher": 16.14}
|
| 44 |
+
{"step": 44, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.055975045030827945, "tokens": 120000, "cumulative_loss_tokens": 5280000, "grad_norm": 0.3515625, "lr": 3e-05, "finish_rate": 0.748, "comp_len": 594.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.474, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 85.4, "frames": {"chat": 202}, "mem_gb": 16.03, "mem_gb_teacher": 16.03}
|
| 45 |
+
{"step": 45, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06047188884726105, "tokens": 120000, "cumulative_loss_tokens": 5400000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.36, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.3, "frames": {"chat": 217}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 46 |
+
{"step": 46, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05655016646341731, "tokens": 120000, "cumulative_loss_tokens": 5520000, "grad_norm": 0.357421875, "lr": 3e-05, "finish_rate": 0.866, "comp_len": 535.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.47, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 92.3, "frames": {"chat": 224}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 47 |
+
{"step": 47, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0790889122961089, "tokens": 120000, "cumulative_loss_tokens": 5640000, "grad_norm": 0.828125, "lr": 3e-05, "finish_rate": 0.753, "comp_len": 558.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.285, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.5, "frames": {"chat": 215}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 48 |
+
{"step": 48, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05377363468687981, "tokens": 120000, "cumulative_loss_tokens": 5760000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 463.3, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.356, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 104.3, "frames": {"chat": 259}, "mem_gb": 15.97, "mem_gb_teacher": 15.97}
|
| 49 |
+
{"step": 49, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.06014201375305032, "tokens": 120000, "cumulative_loss_tokens": 5880000, "grad_norm": 0.3359375, "lr": 3e-05, "finish_rate": 0.829, "comp_len": 571.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.475, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 86.9, "frames": {"chat": 210}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 50 |
+
{"step": 50, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.07395202046850076, "tokens": 120000, "cumulative_loss_tokens": 6000000, "grad_norm": 0.416015625, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.304, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.3, "frames": {"chat": 213}, "mem_gb": 16.09, "mem_gb_teacher": 16.09}
|
| 51 |
+
{"step": 51, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.05859705059945894, "tokens": 120000, "cumulative_loss_tokens": 6120000, "grad_norm": 0.34765625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 540.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.453, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.8, "frames": {"chat": 222}, "mem_gb": 16.0, "mem_gb_teacher": 16.0}
|
| 52 |
+
{"step": 52, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0539097297622667, "tokens": 120000, "cumulative_loss_tokens": 6240000, "grad_norm": 0.353515625, "lr": 3e-05, "finish_rate": 0.889, "comp_len": 510.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.439, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 96.1, "frames": {"chat": 235}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 53 |
+
{"step": 53, "epoch": 0, "training_mode": "off-policy", "forward_topk_kl": 0.0594748610290233, "tokens": 120000, "cumulative_loss_tokens": 6360000, "grad_norm": 0.328125, "lr": 3e-05, "finish_rate": 0.798, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.417, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 86.8, "frames": {"chat": 208}, "mem_gb": 16.01, "mem_gb_teacher": 16.01}
|
| 54 |
+
{"step": 54, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.045099887475029875, "tokens": 120000, "cumulative_loss_tokens": 6480000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.733, "comp_len": 628.3, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.413, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 80.2, "frames": {"chat": 191}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 55 |
+
{"step": 55, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03569839329215077, "tokens": 120000, "cumulative_loss_tokens": 6600000, "grad_norm": 0.2451171875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 547.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.523, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.0, "frames": {"chat": 219}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 56 |
+
{"step": 56, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04039377661425, "tokens": 120000, "cumulative_loss_tokens": 6720000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.778, "comp_len": 579.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.459, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 86.0, "frames": {"chat": 207}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 57 |
+
{"step": 57, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.055602129854184265, "tokens": 120000, "cumulative_loss_tokens": 6840000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.755, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.449, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 88.0, "frames": {"chat": 208}, "mem_gb": 16.01, "mem_gb_teacher": 16.01}
|
| 58 |
+
{"step": 58, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.035877808295966436, "tokens": 120000, "cumulative_loss_tokens": 6960000, "grad_norm": 0.279296875, "lr": 3e-05, "finish_rate": 0.799, "comp_len": 547.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.288, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.8, "frames": {"chat": 219}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 59 |
+
{"step": 59, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03478414489501932, "tokens": 120000, "cumulative_loss_tokens": 7080000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.915, "comp_len": 487.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.329, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 99.7, "frames": {"chat": 246}, "mem_gb": 15.92, "mem_gb_teacher": 15.92}
|
| 60 |
+
{"step": 60, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04739725781680706, "tokens": 120000, "cumulative_loss_tokens": 7200000, "grad_norm": 0.330078125, "lr": 3e-05, "finish_rate": 0.704, "comp_len": 582.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.437, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 87.5, "frames": {"chat": 206}, "mem_gb": 16.07, "mem_gb_teacher": 16.07}
|
| 61 |
+
{"step": 61, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03820183755345643, "tokens": 120000, "cumulative_loss_tokens": 7320000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 515.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.375, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 95.5, "frames": {"chat": 233}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 62 |
+
{"step": 62, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04388709897000032, "tokens": 120000, "cumulative_loss_tokens": 7440000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.847, "comp_len": 524.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.363, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 93.8, "frames": {"chat": 229}, "mem_gb": 15.92, "mem_gb_teacher": 15.92}
|
| 63 |
+
{"step": 63, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.032643947703313705, "tokens": 120000, "cumulative_loss_tokens": 7560000, "grad_norm": 0.236328125, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 508.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.285, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 96.4, "frames": {"chat": 236}, "mem_gb": 15.95, "mem_gb_teacher": 15.95}
|
| 64 |
+
{"step": 64, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04086883016227123, "tokens": 120000, "cumulative_loss_tokens": 7680000, "grad_norm": 0.2451171875, "lr": 3e-05, "finish_rate": 0.87, "comp_len": 502.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.37, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 98.3, "frames": {"chat": 239}, "mem_gb": 15.84, "mem_gb_teacher": 15.84}
|
| 65 |
+
{"step": 65, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03435394472128246, "tokens": 120000, "cumulative_loss_tokens": 7800000, "grad_norm": 0.2177734375, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 497.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.349, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 99.1, "frames": {"chat": 241}, "mem_gb": 15.96, "mem_gb_teacher": 15.96}
|
| 66 |
+
{"step": 66, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.036759108127855385, "tokens": 120000, "cumulative_loss_tokens": 7920000, "grad_norm": 0.2294921875, "lr": 3e-05, "finish_rate": 0.863, "comp_len": 531.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.44, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 93.4, "frames": {"chat": 226}, "mem_gb": 15.92, "mem_gb_teacher": 15.92}
|
| 67 |
+
{"step": 67, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.031785604377323765, "tokens": 120000, "cumulative_loss_tokens": 8040000, "grad_norm": 0.224609375, "lr": 3e-05, "finish_rate": 0.893, "comp_len": 512.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.32, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 95.3, "frames": {"chat": 234}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 68 |
+
{"step": 68, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03207274362700991, "tokens": 120000, "cumulative_loss_tokens": 8160000, "grad_norm": 0.22265625, "lr": 3e-05, "finish_rate": 0.914, "comp_len": 466.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.401, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 102.7, "frames": {"chat": 257}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 69 |
+
{"step": 69, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04908341010484534, "tokens": 120000, "cumulative_loss_tokens": 8280000, "grad_norm": 0.2890625, "lr": 3e-05, "finish_rate": 0.76, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.523, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.3, "frames": {"chat": 208}, "mem_gb": 16.1, "mem_gb_teacher": 16.1}
|
| 70 |
+
{"step": 70, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.043880132612407516, "tokens": 120000, "cumulative_loss_tokens": 8400000, "grad_norm": 0.267578125, "lr": 3e-05, "finish_rate": 0.763, "comp_len": 568.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.522, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.2, "frames": {"chat": 211}, "mem_gb": 16.07, "mem_gb_teacher": 16.07}
|
| 71 |
+
{"step": 71, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04658220293604148, "tokens": 120000, "cumulative_loss_tokens": 8520000, "grad_norm": 0.302734375, "lr": 3e-05, "finish_rate": 0.806, "comp_len": 528.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.407, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 93.8, "frames": {"chat": 227}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 72 |
+
{"step": 72, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04197514075435077, "tokens": 120000, "cumulative_loss_tokens": 8640000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.796, "comp_len": 568.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.434, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.1, "frames": {"chat": 211}, "mem_gb": 16.03, "mem_gb_teacher": 16.03}
|
| 73 |
+
{"step": 73, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03433373855294194, "tokens": 120000, "cumulative_loss_tokens": 8760000, "grad_norm": 0.248046875, "lr": 3e-05, "finish_rate": 0.861, "comp_len": 504.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.403, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 98.0, "frames": {"chat": 238}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 74 |
+
{"step": 74, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.035312130354298275, "tokens": 120000, "cumulative_loss_tokens": 8880000, "grad_norm": 0.2255859375, "lr": 3e-05, "finish_rate": 0.835, "comp_len": 506.3, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.517, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 98.2, "frames": {"chat": 237}, "mem_gb": 16.08, "mem_gb_teacher": 16.08}
|
| 75 |
+
{"step": 75, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04418459653495035, "tokens": 120000, "cumulative_loss_tokens": 9000000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.721, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.339, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 88.6, "frames": {"chat": 208}, "mem_gb": 16.08, "mem_gb_teacher": 16.08}
|
| 76 |
+
{"step": 76, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03462567985369048, "tokens": 120000, "cumulative_loss_tokens": 9120000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.801, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.588, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 91.3, "frames": {"chat": 221}, "mem_gb": 16.17, "mem_gb_teacher": 16.17}
|
| 77 |
+
{"step": 77, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0355419263230792, "tokens": 120000, "cumulative_loss_tokens": 9240000, "grad_norm": 0.244140625, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 517.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.479, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 96.6, "frames": {"chat": 232}, "mem_gb": 16.01, "mem_gb_teacher": 16.01}
|
| 78 |
+
{"step": 78, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03847126886160113, "tokens": 120000, "cumulative_loss_tokens": 9360000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.43, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 87.7, "frames": {"chat": 208}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 79 |
+
{"step": 79, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03187043257508582, "tokens": 120000, "cumulative_loss_tokens": 9480000, "grad_norm": 0.232421875, "lr": 3e-05, "finish_rate": 0.837, "comp_len": 528.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.356, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 93.2, "frames": {"chat": 227}, "mem_gb": 15.96, "mem_gb_teacher": 15.96}
|
| 80 |
+
{"step": 80, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03652716889477645, "tokens": 120000, "cumulative_loss_tokens": 9600000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.824, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.387, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 91.2, "frames": {"chat": 221}, "mem_gb": 15.99, "mem_gb_teacher": 15.99}
|
| 81 |
+
{"step": 81, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03261745297779174, "tokens": 120000, "cumulative_loss_tokens": 9720000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 517.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.457, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 95.5, "frames": {"chat": 232}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 82 |
+
{"step": 82, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03819663266551991, "tokens": 120000, "cumulative_loss_tokens": 9840000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.822, "comp_len": 547.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.423, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.2, "frames": {"chat": 219}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 83 |
+
{"step": 83, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03785421463410991, "tokens": 120000, "cumulative_loss_tokens": 9960000, "grad_norm": 0.2265625, "lr": 3e-05, "finish_rate": 0.713, "comp_len": 615.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.399, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 82.7, "frames": {"chat": 195}, "mem_gb": 16.14, "mem_gb_teacher": 16.14}
|
| 84 |
+
{"step": 84, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03711687127229913, "tokens": 120000, "cumulative_loss_tokens": 10080000, "grad_norm": 0.244140625, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 555.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.503, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 88.9, "frames": {"chat": 216}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 85 |
+
{"step": 85, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.042023429202785095, "tokens": 120000, "cumulative_loss_tokens": 10200000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.788, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.37, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 87.7, "frames": {"chat": 208}, "mem_gb": 15.93, "mem_gb_teacher": 15.93}
|
| 86 |
+
{"step": 86, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03288566896258077, "tokens": 120000, "cumulative_loss_tokens": 10320000, "grad_norm": 0.2265625, "lr": 3e-05, "finish_rate": 0.919, "comp_len": 510.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.407, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 95.6, "frames": {"chat": 235}, "mem_gb": 15.93, "mem_gb_teacher": 15.93}
|
| 87 |
+
{"step": 87, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.033694713056855834, "tokens": 120000, "cumulative_loss_tokens": 10440000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.853, "comp_len": 533.3, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.391, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 91.6, "frames": {"chat": 225}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 88 |
+
{"step": 88, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04422600561644261, "tokens": 120000, "cumulative_loss_tokens": 10560000, "grad_norm": 0.2451171875, "lr": 3e-05, "finish_rate": 0.77, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.483, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.2, "frames": {"chat": 213}, "mem_gb": 16.13, "mem_gb_teacher": 16.13}
|
| 89 |
+
{"step": 89, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.029030231156169126, "tokens": 120000, "cumulative_loss_tokens": 10680000, "grad_norm": 0.2255859375, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 466.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.372, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 104.0, "frames": {"chat": 257}, "mem_gb": 15.8, "mem_gb_teacher": 15.8}
|
| 90 |
+
{"step": 90, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.043001266495510934, "tokens": 120000, "cumulative_loss_tokens": 10800000, "grad_norm": 0.2578125, "lr": 3e-05, "finish_rate": 0.792, "comp_len": 566.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.497, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.1, "frames": {"chat": 212}, "mem_gb": 16.07, "mem_gb_teacher": 16.07}
|
| 91 |
+
{"step": 91, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03456530287990657, "tokens": 120000, "cumulative_loss_tokens": 10920000, "grad_norm": 0.263671875, "lr": 3e-05, "finish_rate": 0.833, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.338, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.9, "frames": {"chat": 221}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 92 |
+
{"step": 92, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.030914492412268495, "tokens": 120000, "cumulative_loss_tokens": 11040000, "grad_norm": 0.203125, "lr": 3e-05, "finish_rate": 0.868, "comp_len": 495.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.442, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 98.2, "frames": {"chat": 242}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 93 |
+
{"step": 93, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0329031198489829, "tokens": 120000, "cumulative_loss_tokens": 11160000, "grad_norm": 0.2216796875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 545.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.352, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 91.5, "frames": {"chat": 220}, "mem_gb": 16.01, "mem_gb_teacher": 16.01}
|
| 94 |
+
{"step": 94, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.029550562638859263, "tokens": 120000, "cumulative_loss_tokens": 11280000, "grad_norm": 0.2197265625, "lr": 3e-05, "finish_rate": 0.896, "comp_len": 500.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.291, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 96.4, "frames": {"chat": 240}, "mem_gb": 15.9, "mem_gb_teacher": 15.9}
|
| 95 |
+
{"step": 95, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0336006456746875, "tokens": 120000, "cumulative_loss_tokens": 11400000, "grad_norm": 0.23046875, "lr": 3e-05, "finish_rate": 0.728, "comp_len": 582.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.359, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 87.0, "frames": {"chat": 206}, "mem_gb": 16.03, "mem_gb_teacher": 16.03}
|
| 96 |
+
{"step": 96, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04226834708025368, "tokens": 120000, "cumulative_loss_tokens": 11520000, "grad_norm": 0.255859375, "lr": 3e-05, "finish_rate": 0.867, "comp_len": 531.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.402, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 92.9, "frames": {"chat": 226}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 97 |
+
{"step": 97, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.054896473065484314, "tokens": 120000, "cumulative_loss_tokens": 11640000, "grad_norm": 0.3046875, "lr": 3e-05, "finish_rate": 0.877, "comp_len": 491.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.382, "t_data_s": 0.1, "t_rollout_s": 0.0, "t_step_s": 99.8, "frames": {"chat": 244}, "mem_gb": 15.83, "mem_gb_teacher": 15.83}
|
| 98 |
+
{"step": 98, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.040829892701484884, "tokens": 120000, "cumulative_loss_tokens": 11760000, "grad_norm": 0.271484375, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 535.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.406, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 92.6, "frames": {"chat": 224}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 99 |
+
{"step": 99, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.031381215056039705, "tokens": 120000, "cumulative_loss_tokens": 11880000, "grad_norm": 0.240234375, "lr": 3e-05, "finish_rate": 0.923, "comp_len": 442.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.319, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 109.0, "frames": {"chat": 271}, "mem_gb": 15.78, "mem_gb_teacher": 15.78}
|
| 100 |
+
{"step": 100, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.036837965581729075, "tokens": 120000, "cumulative_loss_tokens": 12000000, "grad_norm": 0.265625, "lr": 3e-05, "finish_rate": 0.856, "comp_len": 508.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.362, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 96.5, "frames": {"chat": 236}, "mem_gb": 16.06, "mem_gb_teacher": 16.06}
|
| 101 |
+
{"step": 101, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03681235469069021, "tokens": 120000, "cumulative_loss_tokens": 12120000, "grad_norm": 0.244140625, "lr": 3e-05, "finish_rate": 0.841, "comp_len": 517.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.306, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 96.0, "frames": {"chat": 232}, "mem_gb": 15.93, "mem_gb_teacher": 15.93}
|
| 102 |
+
{"step": 102, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03435125927827321, "tokens": 120000, "cumulative_loss_tokens": 12240000, "grad_norm": 0.251953125, "lr": 3e-05, "finish_rate": 0.79, "comp_len": 571.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.392, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 87.6, "frames": {"chat": 210}, "mem_gb": 15.98, "mem_gb_teacher": 15.98}
|
| 103 |
+
{"step": 103, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.033812837561887375, "tokens": 120000, "cumulative_loss_tokens": 12360000, "grad_norm": 0.275390625, "lr": 3e-05, "finish_rate": 0.811, "comp_len": 553.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.431, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.9, "frames": {"chat": 217}, "mem_gb": 15.95, "mem_gb_teacher": 15.95}
|
| 104 |
+
{"step": 104, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.03568466838332825, "tokens": 120000, "cumulative_loss_tokens": 12480000, "grad_norm": 0.25390625, "lr": 3e-05, "finish_rate": 0.839, "comp_len": 535.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.38, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 92.7, "frames": {"chat": 224}, "mem_gb": 16.07, "mem_gb_teacher": 16.07}
|
| 105 |
+
{"step": 105, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.04380085541328881, "tokens": 120000, "cumulative_loss_tokens": 12600000, "grad_norm": 0.287109375, "lr": 3e-05, "finish_rate": 0.749, "comp_len": 591.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.481, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 86.5, "frames": {"chat": 203}, "mem_gb": 15.92, "mem_gb_teacher": 15.92}
|
| 106 |
+
{"step": 106, "epoch": 1, "training_mode": "off-policy", "forward_topk_kl": 0.0331037232719129, "tokens": 120000, "cumulative_loss_tokens": 12720000, "grad_norm": 0.2197265625, "lr": 3e-05, "finish_rate": 0.887, "comp_len": 502.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.326, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 97.2, "frames": {"chat": 239}, "mem_gb": 16.02, "mem_gb_teacher": 16.02}
|
| 107 |
+
{"step": 107, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022136363052297384, "tokens": 120000, "cumulative_loss_tokens": 12840000, "grad_norm": 0.1875, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 472.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.301, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 102.8, "frames": {"chat": 254}, "mem_gb": 15.93, "mem_gb_teacher": 15.93}
|
| 108 |
+
{"step": 108, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02345696447007358, "tokens": 120000, "cumulative_loss_tokens": 12960000, "grad_norm": 0.21875, "lr": 3e-05, "finish_rate": 0.876, "comp_len": 497.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.421, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 98.5, "frames": {"chat": 241}, "mem_gb": 16.03, "mem_gb_teacher": 16.03}
|
| 109 |
+
{"step": 109, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03220615579979494, "tokens": 120000, "cumulative_loss_tokens": 13080000, "grad_norm": 0.2294921875, "lr": 3e-05, "finish_rate": 0.746, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.419, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.1, "frames": {"chat": 213}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 110 |
+
{"step": 110, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.028369275209781095, "tokens": 120000, "cumulative_loss_tokens": 13200000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.864, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.641, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 91.4, "frames": {"chat": 221}, "mem_gb": 16.1, "mem_gb_teacher": 16.1}
|
| 111 |
+
{"step": 111, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.030552633555675855, "tokens": 120000, "cumulative_loss_tokens": 13320000, "grad_norm": 0.201171875, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 612.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.334, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 83.1, "frames": {"chat": 196}, "mem_gb": 16.06, "mem_gb_teacher": 16.06}
|
| 112 |
+
{"step": 112, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02257079966662762, "tokens": 120000, "cumulative_loss_tokens": 13440000, "grad_norm": 0.1689453125, "lr": 3e-05, "finish_rate": 0.926, "comp_len": 444.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.427, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 108.7, "frames": {"chat": 270}, "mem_gb": 15.86, "mem_gb_teacher": 15.86}
|
| 113 |
+
{"step": 113, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024793952927171875, "tokens": 120000, "cumulative_loss_tokens": 13560000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.815, "comp_len": 555.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.327, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.5, "frames": {"chat": 216}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 114 |
+
{"step": 114, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02779970950682958, "tokens": 120000, "cumulative_loss_tokens": 13680000, "grad_norm": 0.1923828125, "lr": 3e-05, "finish_rate": 0.775, "comp_len": 600.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.314, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 83.9, "frames": {"chat": 200}, "mem_gb": 16.01, "mem_gb_teacher": 16.01}
|
| 115 |
+
{"step": 115, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.025435146312182768, "tokens": 120000, "cumulative_loss_tokens": 13800000, "grad_norm": 0.19921875, "lr": 3e-05, "finish_rate": 0.767, "comp_len": 582.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.406, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 86.4, "frames": {"chat": 206}, "mem_gb": 15.96, "mem_gb_teacher": 15.96}
|
| 116 |
+
{"step": 116, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021958818318624982, "tokens": 120000, "cumulative_loss_tokens": 13920000, "grad_norm": 0.1650390625, "lr": 3e-05, "finish_rate": 0.902, "comp_len": 512.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.5, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 94.8, "frames": {"chat": 234}, "mem_gb": 15.99, "mem_gb_teacher": 15.99}
|
| 117 |
+
{"step": 117, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024690080278979926, "tokens": 120000, "cumulative_loss_tokens": 14040000, "grad_norm": 0.171875, "lr": 3e-05, "finish_rate": 0.823, "comp_len": 558.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.325, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 88.4, "frames": {"chat": 215}, "mem_gb": 16.01, "mem_gb_teacher": 16.01}
|
| 118 |
+
{"step": 118, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02221767870101612, "tokens": 120000, "cumulative_loss_tokens": 14160000, "grad_norm": 0.2021484375, "lr": 3e-05, "finish_rate": 0.922, "comp_len": 470.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.447, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 101.5, "frames": {"chat": 255}, "mem_gb": 15.99, "mem_gb_teacher": 15.99}
|
| 119 |
+
{"step": 119, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021592201069470806, "tokens": 120000, "cumulative_loss_tokens": 14280000, "grad_norm": 0.1708984375, "lr": 3e-05, "finish_rate": 0.892, "comp_len": 480.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.377, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 101.8, "frames": {"chat": 250}, "mem_gb": 15.87, "mem_gb_teacher": 15.87}
|
| 120 |
+
{"step": 120, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022036931684900386, "tokens": 120000, "cumulative_loss_tokens": 14400000, "grad_norm": 0.1826171875, "lr": 3e-05, "finish_rate": 0.884, "comp_len": 495.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.525, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 97.8, "frames": {"chat": 242}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 121 |
+
{"step": 121, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.027385966173838823, "tokens": 120000, "cumulative_loss_tokens": 14520000, "grad_norm": 0.177734375, "lr": 3e-05, "finish_rate": 0.729, "comp_len": 603.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.517, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 85.1, "frames": {"chat": 199}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 122 |
+
{"step": 122, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.037581453007894255, "tokens": 120000, "cumulative_loss_tokens": 14640000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.784, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.386, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 88.8, "frames": {"chat": 208}, "mem_gb": 16.08, "mem_gb_teacher": 16.08}
|
| 123 |
+
{"step": 123, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02500725610067602, "tokens": 120000, "cumulative_loss_tokens": 14760000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.764, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.304, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 86.8, "frames": {"chat": 208}, "mem_gb": 16.02, "mem_gb_teacher": 16.02}
|
| 124 |
+
{"step": 124, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.029994825281358013, "tokens": 120000, "cumulative_loss_tokens": 14880000, "grad_norm": 0.2041015625, "lr": 3e-05, "finish_rate": 0.732, "comp_len": 574.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.473, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.1, "frames": {"chat": 209}, "mem_gb": 16.17, "mem_gb_teacher": 16.17}
|
| 125 |
+
{"step": 125, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021332296466493667, "tokens": 120000, "cumulative_loss_tokens": 15000000, "grad_norm": 0.1728515625, "lr": 3e-05, "finish_rate": 0.855, "comp_len": 510.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.321, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 97.2, "frames": {"chat": 235}, "mem_gb": 16.0, "mem_gb_teacher": 16.0}
|
| 126 |
+
{"step": 126, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02423879373523717, "tokens": 120000, "cumulative_loss_tokens": 15120000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.74, "comp_len": 588.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.503, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 85.5, "frames": {"chat": 204}, "mem_gb": 16.0, "mem_gb_teacher": 16.0}
|
| 127 |
+
{"step": 127, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03143483543585365, "tokens": 120000, "cumulative_loss_tokens": 15240000, "grad_norm": 0.24609375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.498, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 88.5, "frames": {"chat": 208}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 128 |
+
{"step": 128, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022677327596287555, "tokens": 120000, "cumulative_loss_tokens": 15360000, "grad_norm": 0.1748046875, "lr": 3e-05, "finish_rate": 0.825, "comp_len": 500.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.456, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 98.6, "frames": {"chat": 240}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 129 |
+
{"step": 129, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02317165856284555, "tokens": 120000, "cumulative_loss_tokens": 15480000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.89, "comp_len": 487.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.425, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 99.3, "frames": {"chat": 246}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 130 |
+
{"step": 130, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02553549518882452, "tokens": 120000, "cumulative_loss_tokens": 15600000, "grad_norm": 0.16796875, "lr": 3e-05, "finish_rate": 0.909, "comp_len": 493.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.299, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 98.1, "frames": {"chat": 243}, "mem_gb": 15.86, "mem_gb_teacher": 15.86}
|
| 131 |
+
{"step": 131, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03025520235665608, "tokens": 120000, "cumulative_loss_tokens": 15720000, "grad_norm": 0.193359375, "lr": 3e-05, "finish_rate": 0.745, "comp_len": 576.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.449, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 87.0, "frames": {"chat": 208}, "mem_gb": 16.06, "mem_gb_teacher": 16.06}
|
| 132 |
+
{"step": 132, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.033440696114464666, "tokens": 120000, "cumulative_loss_tokens": 15840000, "grad_norm": 0.20703125, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 547.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.352, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.4, "frames": {"chat": 219}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 133 |
+
{"step": 133, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.030961406842339785, "tokens": 120000, "cumulative_loss_tokens": 15960000, "grad_norm": 0.205078125, "lr": 3e-05, "finish_rate": 0.782, "comp_len": 568.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.447, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 89.4, "frames": {"chat": 211}, "mem_gb": 16.06, "mem_gb_teacher": 16.06}
|
| 134 |
+
{"step": 134, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.030365196120288845, "tokens": 120000, "cumulative_loss_tokens": 16080000, "grad_norm": 0.25, "lr": 3e-05, "finish_rate": 0.862, "comp_len": 517.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.283, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 95.3, "frames": {"chat": 232}, "mem_gb": 16.02, "mem_gb_teacher": 16.02}
|
| 135 |
+
{"step": 135, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.03637079153128434, "tokens": 120000, "cumulative_loss_tokens": 16200000, "grad_norm": 0.201171875, "lr": 3e-05, "finish_rate": 0.804, "comp_len": 560.7, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.403, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 90.4, "frames": {"chat": 214}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 136 |
+
{"step": 136, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024788761290698312, "tokens": 120000, "cumulative_loss_tokens": 16320000, "grad_norm": 0.173828125, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 531.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.395, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 94.4, "frames": {"chat": 226}, "mem_gb": 15.94, "mem_gb_teacher": 15.94}
|
| 137 |
+
{"step": 137, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02415202263732596, "tokens": 120000, "cumulative_loss_tokens": 16440000, "grad_norm": 0.171875, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 571.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.404, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 88.7, "frames": {"chat": 210}, "mem_gb": 16.06, "mem_gb_teacher": 16.06}
|
| 138 |
+
{"step": 138, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023906889412474507, "tokens": 120000, "cumulative_loss_tokens": 16560000, "grad_norm": 0.1884765625, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 550.5, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.436, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 91.6, "frames": {"chat": 218}, "mem_gb": 15.88, "mem_gb_teacher": 15.88}
|
| 139 |
+
{"step": 139, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02281373731412459, "tokens": 120000, "cumulative_loss_tokens": 16680000, "grad_norm": 0.185546875, "lr": 3e-05, "finish_rate": 0.858, "comp_len": 515.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.354, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 96.0, "frames": {"chat": 233}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 140 |
+
{"step": 140, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.029389094779423128, "tokens": 120000, "cumulative_loss_tokens": 16800000, "grad_norm": 0.1962890625, "lr": 3e-05, "finish_rate": 0.786, "comp_len": 558.1, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.221, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 91.1, "frames": {"chat": 215}, "mem_gb": 16.05, "mem_gb_teacher": 16.05}
|
| 141 |
+
{"step": 141, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02496639485837271, "tokens": 120000, "cumulative_loss_tokens": 16920000, "grad_norm": 0.1826171875, "lr": 3e-05, "finish_rate": 0.845, "comp_len": 515.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.428, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 95.0, "frames": {"chat": 233}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 142 |
+
{"step": 142, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.024896287856064736, "tokens": 120000, "cumulative_loss_tokens": 17040000, "grad_norm": 0.2138671875, "lr": 3e-05, "finish_rate": 0.766, "comp_len": 574.2, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.353, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 87.5, "frames": {"chat": 209}, "mem_gb": 15.99, "mem_gb_teacher": 15.99}
|
| 143 |
+
{"step": 143, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0202058710468933, "tokens": 120000, "cumulative_loss_tokens": 17160000, "grad_norm": 0.166015625, "lr": 3e-05, "finish_rate": 0.908, "comp_len": 458.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.308, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 105.8, "frames": {"chat": 262}, "mem_gb": 15.92, "mem_gb_teacher": 15.92}
|
| 144 |
+
{"step": 144, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.0228134301078273, "tokens": 120000, "cumulative_loss_tokens": 17280000, "grad_norm": 0.171875, "lr": 3e-05, "finish_rate": 0.9, "comp_len": 481.9, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.241, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 101.0, "frames": {"chat": 249}, "mem_gb": 16.01, "mem_gb_teacher": 16.01}
|
| 145 |
+
{"step": 145, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.030403959429240787, "tokens": 120000, "cumulative_loss_tokens": 17400000, "grad_norm": 0.18359375, "lr": 3e-05, "finish_rate": 0.819, "comp_len": 528.6, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.364, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 94.0, "frames": {"chat": 227}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 146 |
+
{"step": 146, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022969909678919553, "tokens": 120000, "cumulative_loss_tokens": 17520000, "grad_norm": 0.1796875, "lr": 3e-05, "finish_rate": 0.814, "comp_len": 543.0, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.356, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 91.2, "frames": {"chat": 221}, "mem_gb": 16.04, "mem_gb_teacher": 16.04}
|
| 147 |
+
{"step": 147, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.02228280872052225, "tokens": 120000, "cumulative_loss_tokens": 17640000, "grad_norm": 0.171875, "lr": 3e-05, "finish_rate": 0.859, "comp_len": 512.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.473, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 95.0, "frames": {"chat": 234}, "mem_gb": 16.06, "mem_gb_teacher": 16.06}
|
| 148 |
+
{"step": 148, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.023852225604599032, "tokens": 120000, "cumulative_loss_tokens": 17760000, "grad_norm": 0.212890625, "lr": 3e-05, "finish_rate": 0.817, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.29, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 87.8, "frames": {"chat": 213}, "mem_gb": 16.0, "mem_gb_teacher": 16.0}
|
| 149 |
+
{"step": 149, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.022865247969034438, "tokens": 120000, "cumulative_loss_tokens": 17880000, "grad_norm": 0.1748046875, "lr": 3e-05, "finish_rate": 0.836, "comp_len": 563.4, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.455, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 88.1, "frames": {"chat": 213}, "mem_gb": 15.94, "mem_gb_teacher": 15.94}
|
| 150 |
+
{"step": 150, "epoch": 2, "training_mode": "off-policy", "forward_topk_kl": 0.021284172641811892, "tokens": 120000, "cumulative_loss_tokens": 18000000, "grad_norm": 0.1669921875, "lr": 3e-05, "finish_rate": 0.906, "comp_len": 512.8, "dropped_truncated": 0, "gold_loss": null, "gold_lambda": null, "rep_ratio": 2.392, "t_data_s": 0.0, "t_rollout_s": 0.0, "t_step_s": 95.1, "frames": {"chat": 234}, "mem_gb": 15.97, "mem_gb_teacher": 15.97}
|
healed/opd_warm_unleashed/args.json
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"student": "outputs/healed/keep50_warmup_fixed_s1224/step0150",
|
| 3 |
+
"teacher": "allenai/OLMoE-1B-7B-0125-Instruct",
|
| 4 |
+
"training_mode": "on-policy",
|
| 5 |
+
"kl_direction": "reverse",
|
| 6 |
+
"dataset": "allenai/Dolci-Instruct-RL",
|
| 7 |
+
"dataset_sources": null,
|
| 8 |
+
"max_difficulty": null,
|
| 9 |
+
"trajectories": "outputs/teacher_trajectories/dolci_math_curated.jsonl",
|
| 10 |
+
"trajectory_dataset": "allenai/Dolci-Instruct-RL",
|
| 11 |
+
"off_policy_frames": "chat",
|
| 12 |
+
"off_policy_max_seq_len": 2048,
|
| 13 |
+
"topk_targets": null,
|
| 14 |
+
"max_loss_tokens": null,
|
| 15 |
+
"loss_tokens_per_step": null,
|
| 16 |
+
"teacher_device": "cuda:0",
|
| 17 |
+
"student_device": "cuda:1",
|
| 18 |
+
"lr": 1e-05,
|
| 19 |
+
"optimizer": "adamw8bit",
|
| 20 |
+
"weight_decay": 0.1,
|
| 21 |
+
"epochs": 2,
|
| 22 |
+
"prompts_per_step": 256,
|
| 23 |
+
"group_size": 4,
|
| 24 |
+
"rollout_batch": 64,
|
| 25 |
+
"micro_batch": 4,
|
| 26 |
+
"max_new_tokens": 2048,
|
| 27 |
+
"max_prompt_len": 1024,
|
| 28 |
+
"warmup_steps": 10,
|
| 29 |
+
"max_grad_norm": 1.0,
|
| 30 |
+
"eval_every": 10,
|
| 31 |
+
"gsm8k_every": 50,
|
| 32 |
+
"gsm8k_n": 256,
|
| 33 |
+
"gsm8k_batch": 16,
|
| 34 |
+
"gsm8k_max_new_tokens": 1024,
|
| 35 |
+
"gsm8k_frames": "chat",
|
| 36 |
+
"save_every": 50,
|
| 37 |
+
"out_dir": "outputs/healed/opd_warm_unleashed",
|
| 38 |
+
"sweep": 500,
|
| 39 |
+
"wandb": true,
|
| 40 |
+
"wandb_project": "glean-heal",
|
| 41 |
+
"wandb_run_name": "opd-warm-unleashed2-keep50-s1223",
|
| 42 |
+
"wandb_run_id": null,
|
| 43 |
+
"wandb_resume": null,
|
| 44 |
+
"wandb_mode": "offline",
|
| 45 |
+
"no_wandb_sync": false,
|
| 46 |
+
"debug": false,
|
| 47 |
+
"resume_from": "outputs/healed/opd_warm_unleashed/step0300",
|
| 48 |
+
"start_step": 300,
|
| 49 |
+
"no_grad_checkpointing": false,
|
| 50 |
+
"seed": 1223,
|
| 51 |
+
"no_teacher_overlap": false,
|
| 52 |
+
"sync_checkpoints": false,
|
| 53 |
+
"rollout_engine": "vllm",
|
| 54 |
+
"vllm_gpu": "2",
|
| 55 |
+
"vllm_port": 8377,
|
| 56 |
+
"vllm_refresh_every": 1,
|
| 57 |
+
"vllm_serve_bin": "vllm-plugin/.venv/bin/python",
|
| 58 |
+
"vllm_gpu_mem_util": 0.85,
|
| 59 |
+
"liger_loss": true,
|
| 60 |
+
"gold_mix_lambda": 0.25,
|
| 61 |
+
"gold_topk_targets": "outputs/teacher_trajectories/dolci_combined_top128",
|
| 62 |
+
"gold_loss": "ce",
|
| 63 |
+
"gold_mix_decay": 0.005,
|
| 64 |
+
"fast_teacher": false,
|
| 65 |
+
"reference_kl_beta": 0.0,
|
| 66 |
+
"drop_truncated_rollouts": false,
|
| 67 |
+
"vllm_max_model_len": null,
|
| 68 |
+
"vllm_refresh_mode": "reload",
|
| 69 |
+
"vllm_live_dir": null,
|
| 70 |
+
"resolved_kl_direction": "reverse"
|
| 71 |
+
}
|
healed/opd_warm_unleashed/step0450/chat_template.jinja
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{{ bos_token }}{% for message in messages %}{% if message['role'] == 'system' %}{{ '<|system|>
|
| 2 |
+
' + message['content'] + '
|
| 3 |
+
' }}{% elif message['role'] == 'user' %}{{ '<|user|>
|
| 4 |
+
' + message['content'] + '
|
| 5 |
+
' }}{% elif message['role'] == 'assistant' %}{% if not loop.last %}{{ '<|assistant|>
|
| 6 |
+
' + message['content'] + eos_token + '
|
| 7 |
+
' }}{% else %}{{ '<|assistant|>
|
| 8 |
+
' + message['content'] + eos_token }}{% endif %}{% endif %}{% if loop.last and add_generation_prompt %}{{ '<|assistant|>
|
| 9 |
+
' }}{% endif %}{% endfor %}
|
healed/opd_warm_unleashed/step0450/config.json
ADDED
|
@@ -0,0 +1,887 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"PrunedOlmoeForCausalLM"
|
| 4 |
+
],
|
| 5 |
+
"attention_bias": false,
|
| 6 |
+
"attention_dropout": 0.0,
|
| 7 |
+
"auto_map": {
|
| 8 |
+
"AutoConfig": "configuration_pruned_olmoe.PrunedOlmoeConfig",
|
| 9 |
+
"AutoModelForCausalLM": "modeling_pruned_olmoe.PrunedOlmoeForCausalLM"
|
| 10 |
+
},
|
| 11 |
+
"clip_qkv": null,
|
| 12 |
+
"dtype": "bfloat16",
|
| 13 |
+
"eos_token_id": 50279,
|
| 14 |
+
"expert_widths": [
|
| 15 |
+
[
|
| 16 |
+
1024,
|
| 17 |
+
384,
|
| 18 |
+
256,
|
| 19 |
+
256,
|
| 20 |
+
768,
|
| 21 |
+
1024,
|
| 22 |
+
128,
|
| 23 |
+
768,
|
| 24 |
+
384,
|
| 25 |
+
128,
|
| 26 |
+
1024,
|
| 27 |
+
384,
|
| 28 |
+
384,
|
| 29 |
+
128,
|
| 30 |
+
640,
|
| 31 |
+
896,
|
| 32 |
+
896,
|
| 33 |
+
256,
|
| 34 |
+
768,
|
| 35 |
+
128,
|
| 36 |
+
768,
|
| 37 |
+
768,
|
| 38 |
+
768,
|
| 39 |
+
768,
|
| 40 |
+
640,
|
| 41 |
+
896,
|
| 42 |
+
256,
|
| 43 |
+
128,
|
| 44 |
+
384,
|
| 45 |
+
896,
|
| 46 |
+
256,
|
| 47 |
+
768,
|
| 48 |
+
512,
|
| 49 |
+
640,
|
| 50 |
+
768,
|
| 51 |
+
896,
|
| 52 |
+
1024,
|
| 53 |
+
384,
|
| 54 |
+
384,
|
| 55 |
+
512,
|
| 56 |
+
384,
|
| 57 |
+
256,
|
| 58 |
+
768,
|
| 59 |
+
640,
|
| 60 |
+
896,
|
| 61 |
+
896,
|
| 62 |
+
512,
|
| 63 |
+
768,
|
| 64 |
+
512,
|
| 65 |
+
384,
|
| 66 |
+
1024,
|
| 67 |
+
896,
|
| 68 |
+
896,
|
| 69 |
+
768,
|
| 70 |
+
128
|
| 71 |
+
],
|
| 72 |
+
[
|
| 73 |
+
768,
|
| 74 |
+
896,
|
| 75 |
+
896,
|
| 76 |
+
256,
|
| 77 |
+
896,
|
| 78 |
+
384,
|
| 79 |
+
768,
|
| 80 |
+
768,
|
| 81 |
+
1024,
|
| 82 |
+
384,
|
| 83 |
+
512,
|
| 84 |
+
1024,
|
| 85 |
+
768,
|
| 86 |
+
768,
|
| 87 |
+
1024,
|
| 88 |
+
640,
|
| 89 |
+
128,
|
| 90 |
+
768,
|
| 91 |
+
384,
|
| 92 |
+
768,
|
| 93 |
+
128,
|
| 94 |
+
256,
|
| 95 |
+
512,
|
| 96 |
+
384,
|
| 97 |
+
896,
|
| 98 |
+
256,
|
| 99 |
+
1024,
|
| 100 |
+
1024,
|
| 101 |
+
896,
|
| 102 |
+
1024,
|
| 103 |
+
896,
|
| 104 |
+
768,
|
| 105 |
+
256,
|
| 106 |
+
768,
|
| 107 |
+
128,
|
| 108 |
+
1024,
|
| 109 |
+
768,
|
| 110 |
+
768,
|
| 111 |
+
640,
|
| 112 |
+
384,
|
| 113 |
+
1024,
|
| 114 |
+
384,
|
| 115 |
+
640,
|
| 116 |
+
896,
|
| 117 |
+
1024,
|
| 118 |
+
512,
|
| 119 |
+
512,
|
| 120 |
+
512,
|
| 121 |
+
640
|
| 122 |
+
],
|
| 123 |
+
[
|
| 124 |
+
768,
|
| 125 |
+
1024,
|
| 126 |
+
512,
|
| 127 |
+
384,
|
| 128 |
+
384,
|
| 129 |
+
768,
|
| 130 |
+
256,
|
| 131 |
+
384,
|
| 132 |
+
1024,
|
| 133 |
+
896,
|
| 134 |
+
640,
|
| 135 |
+
256,
|
| 136 |
+
1024,
|
| 137 |
+
1024,
|
| 138 |
+
1024,
|
| 139 |
+
640,
|
| 140 |
+
1024,
|
| 141 |
+
896,
|
| 142 |
+
512,
|
| 143 |
+
768,
|
| 144 |
+
128,
|
| 145 |
+
768,
|
| 146 |
+
1024,
|
| 147 |
+
768,
|
| 148 |
+
256,
|
| 149 |
+
1024,
|
| 150 |
+
896,
|
| 151 |
+
1024,
|
| 152 |
+
896,
|
| 153 |
+
512,
|
| 154 |
+
640,
|
| 155 |
+
384,
|
| 156 |
+
896,
|
| 157 |
+
384,
|
| 158 |
+
512,
|
| 159 |
+
640,
|
| 160 |
+
256,
|
| 161 |
+
768,
|
| 162 |
+
256,
|
| 163 |
+
768,
|
| 164 |
+
768,
|
| 165 |
+
640,
|
| 166 |
+
1024,
|
| 167 |
+
640,
|
| 168 |
+
896,
|
| 169 |
+
768,
|
| 170 |
+
768,
|
| 171 |
+
256
|
| 172 |
+
],
|
| 173 |
+
[
|
| 174 |
+
640,
|
| 175 |
+
640,
|
| 176 |
+
896,
|
| 177 |
+
512,
|
| 178 |
+
640,
|
| 179 |
+
896,
|
| 180 |
+
896,
|
| 181 |
+
384,
|
| 182 |
+
896,
|
| 183 |
+
896,
|
| 184 |
+
384,
|
| 185 |
+
896,
|
| 186 |
+
640,
|
| 187 |
+
256,
|
| 188 |
+
640,
|
| 189 |
+
1024,
|
| 190 |
+
1024,
|
| 191 |
+
768,
|
| 192 |
+
1024,
|
| 193 |
+
896,
|
| 194 |
+
1024,
|
| 195 |
+
768,
|
| 196 |
+
896,
|
| 197 |
+
256,
|
| 198 |
+
512,
|
| 199 |
+
768,
|
| 200 |
+
1024,
|
| 201 |
+
256,
|
| 202 |
+
768,
|
| 203 |
+
512,
|
| 204 |
+
256,
|
| 205 |
+
640,
|
| 206 |
+
1024,
|
| 207 |
+
1024,
|
| 208 |
+
512,
|
| 209 |
+
1024,
|
| 210 |
+
768,
|
| 211 |
+
256,
|
| 212 |
+
1024,
|
| 213 |
+
384,
|
| 214 |
+
896,
|
| 215 |
+
1024,
|
| 216 |
+
896,
|
| 217 |
+
1024,
|
| 218 |
+
384
|
| 219 |
+
],
|
| 220 |
+
[
|
| 221 |
+
256,
|
| 222 |
+
640,
|
| 223 |
+
640,
|
| 224 |
+
896,
|
| 225 |
+
768,
|
| 226 |
+
896,
|
| 227 |
+
768,
|
| 228 |
+
768,
|
| 229 |
+
896,
|
| 230 |
+
896,
|
| 231 |
+
1024,
|
| 232 |
+
256,
|
| 233 |
+
512,
|
| 234 |
+
1024,
|
| 235 |
+
640,
|
| 236 |
+
896,
|
| 237 |
+
512,
|
| 238 |
+
512,
|
| 239 |
+
384,
|
| 240 |
+
384,
|
| 241 |
+
256,
|
| 242 |
+
384,
|
| 243 |
+
384,
|
| 244 |
+
896,
|
| 245 |
+
896,
|
| 246 |
+
768,
|
| 247 |
+
640,
|
| 248 |
+
896,
|
| 249 |
+
768,
|
| 250 |
+
1024,
|
| 251 |
+
512,
|
| 252 |
+
640,
|
| 253 |
+
512,
|
| 254 |
+
640,
|
| 255 |
+
896,
|
| 256 |
+
512,
|
| 257 |
+
512,
|
| 258 |
+
384,
|
| 259 |
+
640,
|
| 260 |
+
896,
|
| 261 |
+
896,
|
| 262 |
+
896,
|
| 263 |
+
1024,
|
| 264 |
+
640,
|
| 265 |
+
1024,
|
| 266 |
+
640,
|
| 267 |
+
1024
|
| 268 |
+
],
|
| 269 |
+
[
|
| 270 |
+
1024,
|
| 271 |
+
512,
|
| 272 |
+
1024,
|
| 273 |
+
1024,
|
| 274 |
+
640,
|
| 275 |
+
896,
|
| 276 |
+
640,
|
| 277 |
+
1024,
|
| 278 |
+
896,
|
| 279 |
+
384,
|
| 280 |
+
1024,
|
| 281 |
+
128,
|
| 282 |
+
896,
|
| 283 |
+
768,
|
| 284 |
+
1024,
|
| 285 |
+
768,
|
| 286 |
+
640,
|
| 287 |
+
896,
|
| 288 |
+
768,
|
| 289 |
+
640,
|
| 290 |
+
512,
|
| 291 |
+
896,
|
| 292 |
+
512,
|
| 293 |
+
640,
|
| 294 |
+
256,
|
| 295 |
+
768,
|
| 296 |
+
640,
|
| 297 |
+
768,
|
| 298 |
+
384,
|
| 299 |
+
896,
|
| 300 |
+
512,
|
| 301 |
+
512,
|
| 302 |
+
256,
|
| 303 |
+
512,
|
| 304 |
+
896,
|
| 305 |
+
256,
|
| 306 |
+
384,
|
| 307 |
+
640,
|
| 308 |
+
512,
|
| 309 |
+
640,
|
| 310 |
+
896,
|
| 311 |
+
512,
|
| 312 |
+
1024,
|
| 313 |
+
256,
|
| 314 |
+
768,
|
| 315 |
+
1024,
|
| 316 |
+
768,
|
| 317 |
+
256,
|
| 318 |
+
256
|
| 319 |
+
],
|
| 320 |
+
[
|
| 321 |
+
640,
|
| 322 |
+
896,
|
| 323 |
+
1024,
|
| 324 |
+
896,
|
| 325 |
+
1024,
|
| 326 |
+
1024,
|
| 327 |
+
1024,
|
| 328 |
+
512,
|
| 329 |
+
256,
|
| 330 |
+
256,
|
| 331 |
+
1024,
|
| 332 |
+
768,
|
| 333 |
+
512,
|
| 334 |
+
768,
|
| 335 |
+
1024,
|
| 336 |
+
1024,
|
| 337 |
+
1024,
|
| 338 |
+
384,
|
| 339 |
+
512,
|
| 340 |
+
1024,
|
| 341 |
+
512,
|
| 342 |
+
1024,
|
| 343 |
+
128,
|
| 344 |
+
640,
|
| 345 |
+
640,
|
| 346 |
+
896,
|
| 347 |
+
768,
|
| 348 |
+
128,
|
| 349 |
+
256,
|
| 350 |
+
256,
|
| 351 |
+
256,
|
| 352 |
+
256,
|
| 353 |
+
896,
|
| 354 |
+
1024,
|
| 355 |
+
1024,
|
| 356 |
+
384,
|
| 357 |
+
896,
|
| 358 |
+
256,
|
| 359 |
+
896,
|
| 360 |
+
640,
|
| 361 |
+
1024,
|
| 362 |
+
384,
|
| 363 |
+
640,
|
| 364 |
+
256,
|
| 365 |
+
1024,
|
| 366 |
+
1024,
|
| 367 |
+
1024
|
| 368 |
+
],
|
| 369 |
+
[
|
| 370 |
+
1024,
|
| 371 |
+
384,
|
| 372 |
+
1024,
|
| 373 |
+
1024,
|
| 374 |
+
256,
|
| 375 |
+
128,
|
| 376 |
+
256,
|
| 377 |
+
384,
|
| 378 |
+
256,
|
| 379 |
+
384,
|
| 380 |
+
896,
|
| 381 |
+
768,
|
| 382 |
+
896,
|
| 383 |
+
896,
|
| 384 |
+
512,
|
| 385 |
+
896,
|
| 386 |
+
640,
|
| 387 |
+
384,
|
| 388 |
+
384,
|
| 389 |
+
896,
|
| 390 |
+
768,
|
| 391 |
+
384,
|
| 392 |
+
896,
|
| 393 |
+
768,
|
| 394 |
+
768,
|
| 395 |
+
512,
|
| 396 |
+
896,
|
| 397 |
+
768,
|
| 398 |
+
768,
|
| 399 |
+
896,
|
| 400 |
+
128,
|
| 401 |
+
896,
|
| 402 |
+
512,
|
| 403 |
+
256,
|
| 404 |
+
768,
|
| 405 |
+
128,
|
| 406 |
+
384,
|
| 407 |
+
256,
|
| 408 |
+
896,
|
| 409 |
+
896,
|
| 410 |
+
384,
|
| 411 |
+
768,
|
| 412 |
+
512,
|
| 413 |
+
640,
|
| 414 |
+
256,
|
| 415 |
+
768,
|
| 416 |
+
640,
|
| 417 |
+
896,
|
| 418 |
+
384,
|
| 419 |
+
512,
|
| 420 |
+
1024,
|
| 421 |
+
768,
|
| 422 |
+
384
|
| 423 |
+
],
|
| 424 |
+
[
|
| 425 |
+
512,
|
| 426 |
+
768,
|
| 427 |
+
512,
|
| 428 |
+
256,
|
| 429 |
+
128,
|
| 430 |
+
640,
|
| 431 |
+
384,
|
| 432 |
+
640,
|
| 433 |
+
768,
|
| 434 |
+
896,
|
| 435 |
+
640,
|
| 436 |
+
768,
|
| 437 |
+
256,
|
| 438 |
+
384,
|
| 439 |
+
1024,
|
| 440 |
+
896,
|
| 441 |
+
256,
|
| 442 |
+
896,
|
| 443 |
+
512,
|
| 444 |
+
256,
|
| 445 |
+
896,
|
| 446 |
+
768,
|
| 447 |
+
256,
|
| 448 |
+
896,
|
| 449 |
+
896,
|
| 450 |
+
384,
|
| 451 |
+
896,
|
| 452 |
+
640,
|
| 453 |
+
768,
|
| 454 |
+
512,
|
| 455 |
+
768,
|
| 456 |
+
768,
|
| 457 |
+
1024,
|
| 458 |
+
768,
|
| 459 |
+
640,
|
| 460 |
+
768,
|
| 461 |
+
384,
|
| 462 |
+
256,
|
| 463 |
+
512,
|
| 464 |
+
896,
|
| 465 |
+
128,
|
| 466 |
+
384,
|
| 467 |
+
256,
|
| 468 |
+
768,
|
| 469 |
+
384,
|
| 470 |
+
256,
|
| 471 |
+
1024,
|
| 472 |
+
1024,
|
| 473 |
+
896,
|
| 474 |
+
256,
|
| 475 |
+
1024,
|
| 476 |
+
256,
|
| 477 |
+
128,
|
| 478 |
+
896
|
| 479 |
+
],
|
| 480 |
+
[
|
| 481 |
+
640,
|
| 482 |
+
640,
|
| 483 |
+
896,
|
| 484 |
+
256,
|
| 485 |
+
1024,
|
| 486 |
+
512,
|
| 487 |
+
1024,
|
| 488 |
+
768,
|
| 489 |
+
384,
|
| 490 |
+
512,
|
| 491 |
+
256,
|
| 492 |
+
768,
|
| 493 |
+
896,
|
| 494 |
+
768,
|
| 495 |
+
512,
|
| 496 |
+
768,
|
| 497 |
+
768,
|
| 498 |
+
640,
|
| 499 |
+
384,
|
| 500 |
+
768,
|
| 501 |
+
512,
|
| 502 |
+
768,
|
| 503 |
+
768,
|
| 504 |
+
512,
|
| 505 |
+
768,
|
| 506 |
+
128,
|
| 507 |
+
896,
|
| 508 |
+
512,
|
| 509 |
+
768,
|
| 510 |
+
1024,
|
| 511 |
+
128,
|
| 512 |
+
384,
|
| 513 |
+
768,
|
| 514 |
+
768,
|
| 515 |
+
768,
|
| 516 |
+
384,
|
| 517 |
+
512,
|
| 518 |
+
640,
|
| 519 |
+
768,
|
| 520 |
+
512,
|
| 521 |
+
768,
|
| 522 |
+
1024,
|
| 523 |
+
640,
|
| 524 |
+
896,
|
| 525 |
+
256,
|
| 526 |
+
1024,
|
| 527 |
+
384,
|
| 528 |
+
768,
|
| 529 |
+
768,
|
| 530 |
+
768
|
| 531 |
+
],
|
| 532 |
+
[
|
| 533 |
+
896,
|
| 534 |
+
512,
|
| 535 |
+
896,
|
| 536 |
+
768,
|
| 537 |
+
384,
|
| 538 |
+
384,
|
| 539 |
+
768,
|
| 540 |
+
512,
|
| 541 |
+
768,
|
| 542 |
+
512,
|
| 543 |
+
1024,
|
| 544 |
+
640,
|
| 545 |
+
896,
|
| 546 |
+
896,
|
| 547 |
+
256,
|
| 548 |
+
640,
|
| 549 |
+
1024,
|
| 550 |
+
256,
|
| 551 |
+
896,
|
| 552 |
+
128,
|
| 553 |
+
128,
|
| 554 |
+
128,
|
| 555 |
+
768,
|
| 556 |
+
896,
|
| 557 |
+
384,
|
| 558 |
+
896,
|
| 559 |
+
512,
|
| 560 |
+
896,
|
| 561 |
+
384,
|
| 562 |
+
256,
|
| 563 |
+
640,
|
| 564 |
+
640,
|
| 565 |
+
896,
|
| 566 |
+
768,
|
| 567 |
+
640,
|
| 568 |
+
256,
|
| 569 |
+
896,
|
| 570 |
+
896,
|
| 571 |
+
512,
|
| 572 |
+
128,
|
| 573 |
+
896,
|
| 574 |
+
256,
|
| 575 |
+
256,
|
| 576 |
+
640,
|
| 577 |
+
896,
|
| 578 |
+
896,
|
| 579 |
+
128,
|
| 580 |
+
1024,
|
| 581 |
+
256,
|
| 582 |
+
384,
|
| 583 |
+
1024,
|
| 584 |
+
640,
|
| 585 |
+
896
|
| 586 |
+
],
|
| 587 |
+
[
|
| 588 |
+
768,
|
| 589 |
+
384,
|
| 590 |
+
640,
|
| 591 |
+
896,
|
| 592 |
+
256,
|
| 593 |
+
128,
|
| 594 |
+
384,
|
| 595 |
+
896,
|
| 596 |
+
128,
|
| 597 |
+
128,
|
| 598 |
+
896,
|
| 599 |
+
256,
|
| 600 |
+
384,
|
| 601 |
+
896,
|
| 602 |
+
512,
|
| 603 |
+
768,
|
| 604 |
+
768,
|
| 605 |
+
512,
|
| 606 |
+
512,
|
| 607 |
+
768,
|
| 608 |
+
896,
|
| 609 |
+
640,
|
| 610 |
+
768,
|
| 611 |
+
896,
|
| 612 |
+
896,
|
| 613 |
+
896,
|
| 614 |
+
640,
|
| 615 |
+
896,
|
| 616 |
+
640,
|
| 617 |
+
512,
|
| 618 |
+
896,
|
| 619 |
+
256,
|
| 620 |
+
512,
|
| 621 |
+
128,
|
| 622 |
+
512,
|
| 623 |
+
384,
|
| 624 |
+
768,
|
| 625 |
+
768,
|
| 626 |
+
1024,
|
| 627 |
+
256,
|
| 628 |
+
768,
|
| 629 |
+
256,
|
| 630 |
+
768,
|
| 631 |
+
512,
|
| 632 |
+
640,
|
| 633 |
+
1024,
|
| 634 |
+
128,
|
| 635 |
+
896,
|
| 636 |
+
896,
|
| 637 |
+
896,
|
| 638 |
+
896,
|
| 639 |
+
1024
|
| 640 |
+
],
|
| 641 |
+
[
|
| 642 |
+
512,
|
| 643 |
+
384,
|
| 644 |
+
768,
|
| 645 |
+
640,
|
| 646 |
+
640,
|
| 647 |
+
768,
|
| 648 |
+
1024,
|
| 649 |
+
896,
|
| 650 |
+
512,
|
| 651 |
+
256,
|
| 652 |
+
640,
|
| 653 |
+
768,
|
| 654 |
+
640,
|
| 655 |
+
896,
|
| 656 |
+
128,
|
| 657 |
+
256,
|
| 658 |
+
896,
|
| 659 |
+
1024,
|
| 660 |
+
256,
|
| 661 |
+
640,
|
| 662 |
+
512,
|
| 663 |
+
256,
|
| 664 |
+
128,
|
| 665 |
+
512,
|
| 666 |
+
256,
|
| 667 |
+
640,
|
| 668 |
+
768,
|
| 669 |
+
768,
|
| 670 |
+
128,
|
| 671 |
+
128,
|
| 672 |
+
768,
|
| 673 |
+
640,
|
| 674 |
+
1024,
|
| 675 |
+
1024,
|
| 676 |
+
768,
|
| 677 |
+
512,
|
| 678 |
+
896,
|
| 679 |
+
768,
|
| 680 |
+
896,
|
| 681 |
+
1024,
|
| 682 |
+
896,
|
| 683 |
+
896,
|
| 684 |
+
512,
|
| 685 |
+
640,
|
| 686 |
+
1024,
|
| 687 |
+
512,
|
| 688 |
+
1024,
|
| 689 |
+
512,
|
| 690 |
+
512,
|
| 691 |
+
512,
|
| 692 |
+
768
|
| 693 |
+
],
|
| 694 |
+
[
|
| 695 |
+
896,
|
| 696 |
+
896,
|
| 697 |
+
1024,
|
| 698 |
+
1024,
|
| 699 |
+
896,
|
| 700 |
+
128,
|
| 701 |
+
768,
|
| 702 |
+
256,
|
| 703 |
+
1024,
|
| 704 |
+
256,
|
| 705 |
+
1024,
|
| 706 |
+
640,
|
| 707 |
+
384,
|
| 708 |
+
256,
|
| 709 |
+
256,
|
| 710 |
+
512,
|
| 711 |
+
768,
|
| 712 |
+
896,
|
| 713 |
+
512,
|
| 714 |
+
768,
|
| 715 |
+
384,
|
| 716 |
+
1024,
|
| 717 |
+
896,
|
| 718 |
+
896,
|
| 719 |
+
1024,
|
| 720 |
+
896,
|
| 721 |
+
768,
|
| 722 |
+
896,
|
| 723 |
+
640,
|
| 724 |
+
1024,
|
| 725 |
+
512,
|
| 726 |
+
896,
|
| 727 |
+
512,
|
| 728 |
+
1024,
|
| 729 |
+
512,
|
| 730 |
+
512,
|
| 731 |
+
256,
|
| 732 |
+
256,
|
| 733 |
+
256,
|
| 734 |
+
512,
|
| 735 |
+
768,
|
| 736 |
+
128,
|
| 737 |
+
384,
|
| 738 |
+
512,
|
| 739 |
+
896,
|
| 740 |
+
896,
|
| 741 |
+
1024,
|
| 742 |
+
256,
|
| 743 |
+
384,
|
| 744 |
+
640
|
| 745 |
+
],
|
| 746 |
+
[
|
| 747 |
+
896,
|
| 748 |
+
640,
|
| 749 |
+
384,
|
| 750 |
+
512,
|
| 751 |
+
256,
|
| 752 |
+
640,
|
| 753 |
+
1024,
|
| 754 |
+
384,
|
| 755 |
+
1024,
|
| 756 |
+
1024,
|
| 757 |
+
768,
|
| 758 |
+
256,
|
| 759 |
+
1024,
|
| 760 |
+
768,
|
| 761 |
+
512,
|
| 762 |
+
896,
|
| 763 |
+
256,
|
| 764 |
+
1024,
|
| 765 |
+
768,
|
| 766 |
+
768,
|
| 767 |
+
768,
|
| 768 |
+
384,
|
| 769 |
+
384,
|
| 770 |
+
256,
|
| 771 |
+
1024,
|
| 772 |
+
384,
|
| 773 |
+
384,
|
| 774 |
+
384,
|
| 775 |
+
896,
|
| 776 |
+
768,
|
| 777 |
+
640,
|
| 778 |
+
768,
|
| 779 |
+
512,
|
| 780 |
+
896,
|
| 781 |
+
896,
|
| 782 |
+
896,
|
| 783 |
+
896,
|
| 784 |
+
256,
|
| 785 |
+
384,
|
| 786 |
+
128,
|
| 787 |
+
1024,
|
| 788 |
+
896,
|
| 789 |
+
256,
|
| 790 |
+
256,
|
| 791 |
+
768,
|
| 792 |
+
640,
|
| 793 |
+
896,
|
| 794 |
+
384,
|
| 795 |
+
768,
|
| 796 |
+
512,
|
| 797 |
+
640
|
| 798 |
+
],
|
| 799 |
+
[
|
| 800 |
+
896,
|
| 801 |
+
1024,
|
| 802 |
+
768,
|
| 803 |
+
1024,
|
| 804 |
+
896,
|
| 805 |
+
256,
|
| 806 |
+
768,
|
| 807 |
+
128,
|
| 808 |
+
128,
|
| 809 |
+
768,
|
| 810 |
+
512,
|
| 811 |
+
896,
|
| 812 |
+
384,
|
| 813 |
+
768,
|
| 814 |
+
1024,
|
| 815 |
+
256,
|
| 816 |
+
768,
|
| 817 |
+
768,
|
| 818 |
+
256,
|
| 819 |
+
512,
|
| 820 |
+
512,
|
| 821 |
+
640,
|
| 822 |
+
512,
|
| 823 |
+
256,
|
| 824 |
+
768,
|
| 825 |
+
896,
|
| 826 |
+
384,
|
| 827 |
+
1024,
|
| 828 |
+
640,
|
| 829 |
+
1024,
|
| 830 |
+
512,
|
| 831 |
+
512,
|
| 832 |
+
384,
|
| 833 |
+
512,
|
| 834 |
+
512,
|
| 835 |
+
1024,
|
| 836 |
+
384,
|
| 837 |
+
896,
|
| 838 |
+
768,
|
| 839 |
+
384,
|
| 840 |
+
384,
|
| 841 |
+
128,
|
| 842 |
+
384,
|
| 843 |
+
1024,
|
| 844 |
+
896,
|
| 845 |
+
640,
|
| 846 |
+
768,
|
| 847 |
+
768,
|
| 848 |
+
256,
|
| 849 |
+
640,
|
| 850 |
+
512,
|
| 851 |
+
640,
|
| 852 |
+
384
|
| 853 |
+
]
|
| 854 |
+
],
|
| 855 |
+
"glean_metadata": {
|
| 856 |
+
"base_model": "allenai/OLMoE-1B-7B-0125-Instruct",
|
| 857 |
+
"block_size": 128,
|
| 858 |
+
"criterion": "reap",
|
| 859 |
+
"dead_experts": 217,
|
| 860 |
+
"keep_fraction": 0.5,
|
| 861 |
+
"min_width": 128,
|
| 862 |
+
"params": 3697491968,
|
| 863 |
+
"scores": "outputs/scores_0125inst_dolmino-math/scores.pt"
|
| 864 |
+
},
|
| 865 |
+
"hidden_act": "silu",
|
| 866 |
+
"hidden_size": 2048,
|
| 867 |
+
"initializer_range": 0.02,
|
| 868 |
+
"intermediate_size": 1024,
|
| 869 |
+
"max_position_embeddings": 4096,
|
| 870 |
+
"model_type": "pruned_olmoe",
|
| 871 |
+
"norm_topk_prob": false,
|
| 872 |
+
"num_attention_heads": 16,
|
| 873 |
+
"num_experts": 64,
|
| 874 |
+
"num_experts_per_tok": 8,
|
| 875 |
+
"num_hidden_layers": 16,
|
| 876 |
+
"num_key_value_heads": 16,
|
| 877 |
+
"output_router_logits": false,
|
| 878 |
+
"pad_token_id": 1,
|
| 879 |
+
"rms_norm_eps": 1e-05,
|
| 880 |
+
"rope_scaling": null,
|
| 881 |
+
"rope_theta": 10000.0,
|
| 882 |
+
"router_aux_loss_coef": 0.01,
|
| 883 |
+
"tie_word_embeddings": false,
|
| 884 |
+
"transformers_version": "4.57.6",
|
| 885 |
+
"use_cache": false,
|
| 886 |
+
"vocab_size": 50304
|
| 887 |
+
}
|
healed/opd_warm_unleashed/step0450/configuration_pruned_olmoe.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Configuration for GLEAN-pruned OLMoE: variable-width, variable-count experts.
|
| 2 |
+
"""
|
| 3 |
+
|
| 4 |
+
from transformers.models.olmoe.configuration_olmoe import OlmoeConfig
|
| 5 |
+
|
| 6 |
+
|
| 7 |
+
class PrunedOlmoeConfig(OlmoeConfig):
|
| 8 |
+
"""OlmoeConfig plus a per-(layer, expert) width table.
|
| 9 |
+
|
| 10 |
+
``expert_widths[l]`` lists the SwiGLU intermediate width of each surviving
|
| 11 |
+
expert in decoder layer ``l``, in expert order. Lists are ragged: layers
|
| 12 |
+
may keep different numbers of experts (deleted experts simply don't
|
| 13 |
+
appear — the router in layer ``l`` has ``len(expert_widths[l])`` rows),
|
| 14 |
+
and each width may differ (multiples of the GEMM block size, 128, for
|
| 15 |
+
variable-MegaBlocks execution). ``None`` means an unpruned model
|
| 16 |
+
(uniform ``num_experts`` × ``intermediate_size``).
|
| 17 |
+
|
| 18 |
+
The inherited ``num_experts`` / ``intermediate_size`` keep their ORIGINAL
|
| 19 |
+
(pre-pruning) values for provenance; the width table is authoritative for
|
| 20 |
+
the built architecture.
|
| 21 |
+
"""
|
| 22 |
+
|
| 23 |
+
model_type = "pruned_olmoe"
|
| 24 |
+
|
| 25 |
+
def __init__(self, expert_widths: list[list[int]] | None = None, **kwargs):
|
| 26 |
+
super().__init__(**kwargs)
|
| 27 |
+
self.expert_widths = expert_widths
|
healed/opd_warm_unleashed/step0450/generation_config.json
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_from_model_config": true,
|
| 3 |
+
"eos_token_id": 50279,
|
| 4 |
+
"pad_token_id": 1,
|
| 5 |
+
"transformers_version": "4.57.6"
|
| 6 |
+
}
|
healed/opd_warm_unleashed/step0450/model.safetensors.index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/opd_warm_unleashed/step0450/modeling_pruned_olmoe.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""GLEAN-pruned OLMoE: HF-loadable model with ragged (variable-width) experts.
|
| 2 |
+
|
| 3 |
+
Pattern follows hbfreed/variable-flex-olmo's PrunedFlexOlmoForCausalLM
|
| 4 |
+
(docs/recon/prior-work-hbfreed.md), generalized from one scalar width to a
|
| 5 |
+
per-(layer, expert) width table: ``super().__init__`` builds the uniform
|
| 6 |
+
architecture from the config, then every MoE block is rebuilt to its pruned
|
| 7 |
+
shape — surviving experts only, each at its own width, router sliced to
|
| 8 |
+
match — so the state dict aligns exactly with what
|
| 9 |
+
``glean.prune.prune_channels_global`` leaves behind.
|
| 10 |
+
|
| 11 |
+
Caveat: ``output_router_logits=True`` (the load-balancing aux loss) assumes a
|
| 12 |
+
uniform ``config.num_experts`` and is unsupported on ragged models.
|
| 13 |
+
"""
|
| 14 |
+
|
| 15 |
+
import torch.nn as nn
|
| 16 |
+
from transformers.activations import ACT2FN
|
| 17 |
+
from transformers.models.olmoe.modeling_olmoe import OlmoeForCausalLM
|
| 18 |
+
|
| 19 |
+
from .configuration_pruned_olmoe import PrunedOlmoeConfig
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
class RaggedOlmoeMLP(nn.Module):
|
| 23 |
+
"""OlmoeMLP with an explicit intermediate width (SwiGLU, no biases)."""
|
| 24 |
+
|
| 25 |
+
def __init__(self, hidden_size: int, intermediate_size: int, hidden_act: str):
|
| 26 |
+
super().__init__()
|
| 27 |
+
self.hidden_size = hidden_size
|
| 28 |
+
self.intermediate_size = intermediate_size
|
| 29 |
+
self.gate_proj = nn.Linear(hidden_size, intermediate_size, bias=False)
|
| 30 |
+
self.up_proj = nn.Linear(hidden_size, intermediate_size, bias=False)
|
| 31 |
+
self.down_proj = nn.Linear(intermediate_size, hidden_size, bias=False)
|
| 32 |
+
self.act_fn = ACT2FN[hidden_act]
|
| 33 |
+
|
| 34 |
+
def forward(self, x):
|
| 35 |
+
return self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 36 |
+
|
| 37 |
+
|
| 38 |
+
class PrunedOlmoeForCausalLM(OlmoeForCausalLM):
|
| 39 |
+
"""OLMoE with per-layer surviving-expert lists at per-expert widths."""
|
| 40 |
+
|
| 41 |
+
config_class = PrunedOlmoeConfig
|
| 42 |
+
|
| 43 |
+
def __init__(self, config: PrunedOlmoeConfig):
|
| 44 |
+
super().__init__(config)
|
| 45 |
+
widths_table = getattr(config, "expert_widths", None)
|
| 46 |
+
if widths_table is None:
|
| 47 |
+
return # unpruned: plain OLMoE
|
| 48 |
+
if len(widths_table) != len(self.model.layers):
|
| 49 |
+
raise ValueError(
|
| 50 |
+
f"expert_widths has {len(widths_table)} rows but the model has "
|
| 51 |
+
f"{len(self.model.layers)} decoder layers"
|
| 52 |
+
)
|
| 53 |
+
for layer, widths in zip(self.model.layers, widths_table):
|
| 54 |
+
if any(w <= 0 for w in widths):
|
| 55 |
+
raise ValueError("expert_widths must list surviving experts only (>0)")
|
| 56 |
+
block = layer.mlp
|
| 57 |
+
if len(widths) < block.top_k:
|
| 58 |
+
raise ValueError(
|
| 59 |
+
f"a layer keeps {len(widths)} experts < top_k={block.top_k}"
|
| 60 |
+
)
|
| 61 |
+
block.num_experts = len(widths)
|
| 62 |
+
block.gate = nn.Linear(config.hidden_size, len(widths), bias=False)
|
| 63 |
+
block.experts = nn.ModuleList(
|
| 64 |
+
RaggedOlmoeMLP(config.hidden_size, w, config.hidden_act)
|
| 65 |
+
for w in widths
|
| 66 |
+
)
|
healed/opd_warm_unleashed/step0450/special_tokens_map.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": {
|
| 3 |
+
"content": "|||IP_ADDRESS|||",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": true,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"eos_token": {
|
| 10 |
+
"content": "|||IP_ADDRESS|||",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": true,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"pad_token": {
|
| 17 |
+
"content": "<pad>",
|
| 18 |
+
"lstrip": false,
|
| 19 |
+
"normalized": false,
|
| 20 |
+
"rstrip": false,
|
| 21 |
+
"single_word": false
|
| 22 |
+
}
|
| 23 |
+
}
|
healed/opd_warm_unleashed/step0450/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/opd_warm_unleashed/step0450/tokenizer_config.json
ADDED
|
@@ -0,0 +1,247 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_bos_token": false,
|
| 3 |
+
"add_eos_token": false,
|
| 4 |
+
"add_prefix_space": false,
|
| 5 |
+
"added_tokens_decoder": {
|
| 6 |
+
"0": {
|
| 7 |
+
"content": "<|endoftext|>",
|
| 8 |
+
"lstrip": false,
|
| 9 |
+
"normalized": false,
|
| 10 |
+
"rstrip": false,
|
| 11 |
+
"single_word": false,
|
| 12 |
+
"special": true
|
| 13 |
+
},
|
| 14 |
+
"1": {
|
| 15 |
+
"content": "<|padding|>",
|
| 16 |
+
"lstrip": false,
|
| 17 |
+
"normalized": false,
|
| 18 |
+
"rstrip": false,
|
| 19 |
+
"single_word": false,
|
| 20 |
+
"special": true
|
| 21 |
+
},
|
| 22 |
+
"50254": {
|
| 23 |
+
"content": " ",
|
| 24 |
+
"lstrip": false,
|
| 25 |
+
"normalized": true,
|
| 26 |
+
"rstrip": false,
|
| 27 |
+
"single_word": false,
|
| 28 |
+
"special": false
|
| 29 |
+
},
|
| 30 |
+
"50255": {
|
| 31 |
+
"content": " ",
|
| 32 |
+
"lstrip": false,
|
| 33 |
+
"normalized": true,
|
| 34 |
+
"rstrip": false,
|
| 35 |
+
"single_word": false,
|
| 36 |
+
"special": false
|
| 37 |
+
},
|
| 38 |
+
"50256": {
|
| 39 |
+
"content": " ",
|
| 40 |
+
"lstrip": false,
|
| 41 |
+
"normalized": true,
|
| 42 |
+
"rstrip": false,
|
| 43 |
+
"single_word": false,
|
| 44 |
+
"special": false
|
| 45 |
+
},
|
| 46 |
+
"50257": {
|
| 47 |
+
"content": " ",
|
| 48 |
+
"lstrip": false,
|
| 49 |
+
"normalized": true,
|
| 50 |
+
"rstrip": false,
|
| 51 |
+
"single_word": false,
|
| 52 |
+
"special": false
|
| 53 |
+
},
|
| 54 |
+
"50258": {
|
| 55 |
+
"content": " ",
|
| 56 |
+
"lstrip": false,
|
| 57 |
+
"normalized": true,
|
| 58 |
+
"rstrip": false,
|
| 59 |
+
"single_word": false,
|
| 60 |
+
"special": false
|
| 61 |
+
},
|
| 62 |
+
"50259": {
|
| 63 |
+
"content": " ",
|
| 64 |
+
"lstrip": false,
|
| 65 |
+
"normalized": true,
|
| 66 |
+
"rstrip": false,
|
| 67 |
+
"single_word": false,
|
| 68 |
+
"special": false
|
| 69 |
+
},
|
| 70 |
+
"50260": {
|
| 71 |
+
"content": " ",
|
| 72 |
+
"lstrip": false,
|
| 73 |
+
"normalized": true,
|
| 74 |
+
"rstrip": false,
|
| 75 |
+
"single_word": false,
|
| 76 |
+
"special": false
|
| 77 |
+
},
|
| 78 |
+
"50261": {
|
| 79 |
+
"content": " ",
|
| 80 |
+
"lstrip": false,
|
| 81 |
+
"normalized": true,
|
| 82 |
+
"rstrip": false,
|
| 83 |
+
"single_word": false,
|
| 84 |
+
"special": false
|
| 85 |
+
},
|
| 86 |
+
"50262": {
|
| 87 |
+
"content": " ",
|
| 88 |
+
"lstrip": false,
|
| 89 |
+
"normalized": true,
|
| 90 |
+
"rstrip": false,
|
| 91 |
+
"single_word": false,
|
| 92 |
+
"special": false
|
| 93 |
+
},
|
| 94 |
+
"50263": {
|
| 95 |
+
"content": " ",
|
| 96 |
+
"lstrip": false,
|
| 97 |
+
"normalized": true,
|
| 98 |
+
"rstrip": false,
|
| 99 |
+
"single_word": false,
|
| 100 |
+
"special": false
|
| 101 |
+
},
|
| 102 |
+
"50264": {
|
| 103 |
+
"content": " ",
|
| 104 |
+
"lstrip": false,
|
| 105 |
+
"normalized": true,
|
| 106 |
+
"rstrip": false,
|
| 107 |
+
"single_word": false,
|
| 108 |
+
"special": false
|
| 109 |
+
},
|
| 110 |
+
"50265": {
|
| 111 |
+
"content": " ",
|
| 112 |
+
"lstrip": false,
|
| 113 |
+
"normalized": true,
|
| 114 |
+
"rstrip": false,
|
| 115 |
+
"single_word": false,
|
| 116 |
+
"special": false
|
| 117 |
+
},
|
| 118 |
+
"50266": {
|
| 119 |
+
"content": " ",
|
| 120 |
+
"lstrip": false,
|
| 121 |
+
"normalized": true,
|
| 122 |
+
"rstrip": false,
|
| 123 |
+
"single_word": false,
|
| 124 |
+
"special": false
|
| 125 |
+
},
|
| 126 |
+
"50267": {
|
| 127 |
+
"content": " ",
|
| 128 |
+
"lstrip": false,
|
| 129 |
+
"normalized": true,
|
| 130 |
+
"rstrip": false,
|
| 131 |
+
"single_word": false,
|
| 132 |
+
"special": false
|
| 133 |
+
},
|
| 134 |
+
"50268": {
|
| 135 |
+
"content": " ",
|
| 136 |
+
"lstrip": false,
|
| 137 |
+
"normalized": true,
|
| 138 |
+
"rstrip": false,
|
| 139 |
+
"single_word": false,
|
| 140 |
+
"special": false
|
| 141 |
+
},
|
| 142 |
+
"50269": {
|
| 143 |
+
"content": " ",
|
| 144 |
+
"lstrip": false,
|
| 145 |
+
"normalized": true,
|
| 146 |
+
"rstrip": false,
|
| 147 |
+
"single_word": false,
|
| 148 |
+
"special": false
|
| 149 |
+
},
|
| 150 |
+
"50270": {
|
| 151 |
+
"content": " ",
|
| 152 |
+
"lstrip": false,
|
| 153 |
+
"normalized": true,
|
| 154 |
+
"rstrip": false,
|
| 155 |
+
"single_word": false,
|
| 156 |
+
"special": false
|
| 157 |
+
},
|
| 158 |
+
"50271": {
|
| 159 |
+
"content": " ",
|
| 160 |
+
"lstrip": false,
|
| 161 |
+
"normalized": true,
|
| 162 |
+
"rstrip": false,
|
| 163 |
+
"single_word": false,
|
| 164 |
+
"special": false
|
| 165 |
+
},
|
| 166 |
+
"50272": {
|
| 167 |
+
"content": " ",
|
| 168 |
+
"lstrip": false,
|
| 169 |
+
"normalized": true,
|
| 170 |
+
"rstrip": false,
|
| 171 |
+
"single_word": false,
|
| 172 |
+
"special": false
|
| 173 |
+
},
|
| 174 |
+
"50273": {
|
| 175 |
+
"content": " ",
|
| 176 |
+
"lstrip": false,
|
| 177 |
+
"normalized": true,
|
| 178 |
+
"rstrip": false,
|
| 179 |
+
"single_word": false,
|
| 180 |
+
"special": false
|
| 181 |
+
},
|
| 182 |
+
"50274": {
|
| 183 |
+
"content": " ",
|
| 184 |
+
"lstrip": false,
|
| 185 |
+
"normalized": true,
|
| 186 |
+
"rstrip": false,
|
| 187 |
+
"single_word": false,
|
| 188 |
+
"special": false
|
| 189 |
+
},
|
| 190 |
+
"50275": {
|
| 191 |
+
"content": " ",
|
| 192 |
+
"lstrip": false,
|
| 193 |
+
"normalized": true,
|
| 194 |
+
"rstrip": false,
|
| 195 |
+
"single_word": false,
|
| 196 |
+
"special": false
|
| 197 |
+
},
|
| 198 |
+
"50276": {
|
| 199 |
+
"content": " ",
|
| 200 |
+
"lstrip": false,
|
| 201 |
+
"normalized": true,
|
| 202 |
+
"rstrip": false,
|
| 203 |
+
"single_word": false,
|
| 204 |
+
"special": false
|
| 205 |
+
},
|
| 206 |
+
"50277": {
|
| 207 |
+
"content": "|||EMAIL_ADDRESS|||",
|
| 208 |
+
"lstrip": false,
|
| 209 |
+
"normalized": true,
|
| 210 |
+
"rstrip": false,
|
| 211 |
+
"single_word": false,
|
| 212 |
+
"special": false
|
| 213 |
+
},
|
| 214 |
+
"50278": {
|
| 215 |
+
"content": "|||PHONE_NUMBER|||",
|
| 216 |
+
"lstrip": false,
|
| 217 |
+
"normalized": true,
|
| 218 |
+
"rstrip": false,
|
| 219 |
+
"single_word": false,
|
| 220 |
+
"special": false
|
| 221 |
+
},
|
| 222 |
+
"50279": {
|
| 223 |
+
"content": "|||IP_ADDRESS|||",
|
| 224 |
+
"lstrip": false,
|
| 225 |
+
"normalized": true,
|
| 226 |
+
"rstrip": false,
|
| 227 |
+
"single_word": false,
|
| 228 |
+
"special": true
|
| 229 |
+
},
|
| 230 |
+
"50280": {
|
| 231 |
+
"content": "<pad>",
|
| 232 |
+
"lstrip": false,
|
| 233 |
+
"normalized": false,
|
| 234 |
+
"rstrip": false,
|
| 235 |
+
"single_word": false,
|
| 236 |
+
"special": true
|
| 237 |
+
}
|
| 238 |
+
},
|
| 239 |
+
"bos_token": "|||IP_ADDRESS|||",
|
| 240 |
+
"clean_up_tokenization_spaces": false,
|
| 241 |
+
"eos_token": "|||IP_ADDRESS|||",
|
| 242 |
+
"extra_special_tokens": {},
|
| 243 |
+
"model_max_length": 1000000000000000019884624838656,
|
| 244 |
+
"pad_token": "<pad>",
|
| 245 |
+
"tokenizer_class": "GPTNeoXTokenizer",
|
| 246 |
+
"unk_token": null
|
| 247 |
+
}
|
healed/opd_warm_unleashed/step0500/chat_template.jinja
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{{ bos_token }}{% for message in messages %}{% if message['role'] == 'system' %}{{ '<|system|>
|
| 2 |
+
' + message['content'] + '
|
| 3 |
+
' }}{% elif message['role'] == 'user' %}{{ '<|user|>
|
| 4 |
+
' + message['content'] + '
|
| 5 |
+
' }}{% elif message['role'] == 'assistant' %}{% if not loop.last %}{{ '<|assistant|>
|
| 6 |
+
' + message['content'] + eos_token + '
|
| 7 |
+
' }}{% else %}{{ '<|assistant|>
|
| 8 |
+
' + message['content'] + eos_token }}{% endif %}{% endif %}{% if loop.last and add_generation_prompt %}{{ '<|assistant|>
|
| 9 |
+
' }}{% endif %}{% endfor %}
|
healed/opd_warm_unleashed/step0500/config.json
ADDED
|
@@ -0,0 +1,887 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"PrunedOlmoeForCausalLM"
|
| 4 |
+
],
|
| 5 |
+
"attention_bias": false,
|
| 6 |
+
"attention_dropout": 0.0,
|
| 7 |
+
"auto_map": {
|
| 8 |
+
"AutoConfig": "configuration_pruned_olmoe.PrunedOlmoeConfig",
|
| 9 |
+
"AutoModelForCausalLM": "modeling_pruned_olmoe.PrunedOlmoeForCausalLM"
|
| 10 |
+
},
|
| 11 |
+
"clip_qkv": null,
|
| 12 |
+
"dtype": "bfloat16",
|
| 13 |
+
"eos_token_id": 50279,
|
| 14 |
+
"expert_widths": [
|
| 15 |
+
[
|
| 16 |
+
1024,
|
| 17 |
+
384,
|
| 18 |
+
256,
|
| 19 |
+
256,
|
| 20 |
+
768,
|
| 21 |
+
1024,
|
| 22 |
+
128,
|
| 23 |
+
768,
|
| 24 |
+
384,
|
| 25 |
+
128,
|
| 26 |
+
1024,
|
| 27 |
+
384,
|
| 28 |
+
384,
|
| 29 |
+
128,
|
| 30 |
+
640,
|
| 31 |
+
896,
|
| 32 |
+
896,
|
| 33 |
+
256,
|
| 34 |
+
768,
|
| 35 |
+
128,
|
| 36 |
+
768,
|
| 37 |
+
768,
|
| 38 |
+
768,
|
| 39 |
+
768,
|
| 40 |
+
640,
|
| 41 |
+
896,
|
| 42 |
+
256,
|
| 43 |
+
128,
|
| 44 |
+
384,
|
| 45 |
+
896,
|
| 46 |
+
256,
|
| 47 |
+
768,
|
| 48 |
+
512,
|
| 49 |
+
640,
|
| 50 |
+
768,
|
| 51 |
+
896,
|
| 52 |
+
1024,
|
| 53 |
+
384,
|
| 54 |
+
384,
|
| 55 |
+
512,
|
| 56 |
+
384,
|
| 57 |
+
256,
|
| 58 |
+
768,
|
| 59 |
+
640,
|
| 60 |
+
896,
|
| 61 |
+
896,
|
| 62 |
+
512,
|
| 63 |
+
768,
|
| 64 |
+
512,
|
| 65 |
+
384,
|
| 66 |
+
1024,
|
| 67 |
+
896,
|
| 68 |
+
896,
|
| 69 |
+
768,
|
| 70 |
+
128
|
| 71 |
+
],
|
| 72 |
+
[
|
| 73 |
+
768,
|
| 74 |
+
896,
|
| 75 |
+
896,
|
| 76 |
+
256,
|
| 77 |
+
896,
|
| 78 |
+
384,
|
| 79 |
+
768,
|
| 80 |
+
768,
|
| 81 |
+
1024,
|
| 82 |
+
384,
|
| 83 |
+
512,
|
| 84 |
+
1024,
|
| 85 |
+
768,
|
| 86 |
+
768,
|
| 87 |
+
1024,
|
| 88 |
+
640,
|
| 89 |
+
128,
|
| 90 |
+
768,
|
| 91 |
+
384,
|
| 92 |
+
768,
|
| 93 |
+
128,
|
| 94 |
+
256,
|
| 95 |
+
512,
|
| 96 |
+
384,
|
| 97 |
+
896,
|
| 98 |
+
256,
|
| 99 |
+
1024,
|
| 100 |
+
1024,
|
| 101 |
+
896,
|
| 102 |
+
1024,
|
| 103 |
+
896,
|
| 104 |
+
768,
|
| 105 |
+
256,
|
| 106 |
+
768,
|
| 107 |
+
128,
|
| 108 |
+
1024,
|
| 109 |
+
768,
|
| 110 |
+
768,
|
| 111 |
+
640,
|
| 112 |
+
384,
|
| 113 |
+
1024,
|
| 114 |
+
384,
|
| 115 |
+
640,
|
| 116 |
+
896,
|
| 117 |
+
1024,
|
| 118 |
+
512,
|
| 119 |
+
512,
|
| 120 |
+
512,
|
| 121 |
+
640
|
| 122 |
+
],
|
| 123 |
+
[
|
| 124 |
+
768,
|
| 125 |
+
1024,
|
| 126 |
+
512,
|
| 127 |
+
384,
|
| 128 |
+
384,
|
| 129 |
+
768,
|
| 130 |
+
256,
|
| 131 |
+
384,
|
| 132 |
+
1024,
|
| 133 |
+
896,
|
| 134 |
+
640,
|
| 135 |
+
256,
|
| 136 |
+
1024,
|
| 137 |
+
1024,
|
| 138 |
+
1024,
|
| 139 |
+
640,
|
| 140 |
+
1024,
|
| 141 |
+
896,
|
| 142 |
+
512,
|
| 143 |
+
768,
|
| 144 |
+
128,
|
| 145 |
+
768,
|
| 146 |
+
1024,
|
| 147 |
+
768,
|
| 148 |
+
256,
|
| 149 |
+
1024,
|
| 150 |
+
896,
|
| 151 |
+
1024,
|
| 152 |
+
896,
|
| 153 |
+
512,
|
| 154 |
+
640,
|
| 155 |
+
384,
|
| 156 |
+
896,
|
| 157 |
+
384,
|
| 158 |
+
512,
|
| 159 |
+
640,
|
| 160 |
+
256,
|
| 161 |
+
768,
|
| 162 |
+
256,
|
| 163 |
+
768,
|
| 164 |
+
768,
|
| 165 |
+
640,
|
| 166 |
+
1024,
|
| 167 |
+
640,
|
| 168 |
+
896,
|
| 169 |
+
768,
|
| 170 |
+
768,
|
| 171 |
+
256
|
| 172 |
+
],
|
| 173 |
+
[
|
| 174 |
+
640,
|
| 175 |
+
640,
|
| 176 |
+
896,
|
| 177 |
+
512,
|
| 178 |
+
640,
|
| 179 |
+
896,
|
| 180 |
+
896,
|
| 181 |
+
384,
|
| 182 |
+
896,
|
| 183 |
+
896,
|
| 184 |
+
384,
|
| 185 |
+
896,
|
| 186 |
+
640,
|
| 187 |
+
256,
|
| 188 |
+
640,
|
| 189 |
+
1024,
|
| 190 |
+
1024,
|
| 191 |
+
768,
|
| 192 |
+
1024,
|
| 193 |
+
896,
|
| 194 |
+
1024,
|
| 195 |
+
768,
|
| 196 |
+
896,
|
| 197 |
+
256,
|
| 198 |
+
512,
|
| 199 |
+
768,
|
| 200 |
+
1024,
|
| 201 |
+
256,
|
| 202 |
+
768,
|
| 203 |
+
512,
|
| 204 |
+
256,
|
| 205 |
+
640,
|
| 206 |
+
1024,
|
| 207 |
+
1024,
|
| 208 |
+
512,
|
| 209 |
+
1024,
|
| 210 |
+
768,
|
| 211 |
+
256,
|
| 212 |
+
1024,
|
| 213 |
+
384,
|
| 214 |
+
896,
|
| 215 |
+
1024,
|
| 216 |
+
896,
|
| 217 |
+
1024,
|
| 218 |
+
384
|
| 219 |
+
],
|
| 220 |
+
[
|
| 221 |
+
256,
|
| 222 |
+
640,
|
| 223 |
+
640,
|
| 224 |
+
896,
|
| 225 |
+
768,
|
| 226 |
+
896,
|
| 227 |
+
768,
|
| 228 |
+
768,
|
| 229 |
+
896,
|
| 230 |
+
896,
|
| 231 |
+
1024,
|
| 232 |
+
256,
|
| 233 |
+
512,
|
| 234 |
+
1024,
|
| 235 |
+
640,
|
| 236 |
+
896,
|
| 237 |
+
512,
|
| 238 |
+
512,
|
| 239 |
+
384,
|
| 240 |
+
384,
|
| 241 |
+
256,
|
| 242 |
+
384,
|
| 243 |
+
384,
|
| 244 |
+
896,
|
| 245 |
+
896,
|
| 246 |
+
768,
|
| 247 |
+
640,
|
| 248 |
+
896,
|
| 249 |
+
768,
|
| 250 |
+
1024,
|
| 251 |
+
512,
|
| 252 |
+
640,
|
| 253 |
+
512,
|
| 254 |
+
640,
|
| 255 |
+
896,
|
| 256 |
+
512,
|
| 257 |
+
512,
|
| 258 |
+
384,
|
| 259 |
+
640,
|
| 260 |
+
896,
|
| 261 |
+
896,
|
| 262 |
+
896,
|
| 263 |
+
1024,
|
| 264 |
+
640,
|
| 265 |
+
1024,
|
| 266 |
+
640,
|
| 267 |
+
1024
|
| 268 |
+
],
|
| 269 |
+
[
|
| 270 |
+
1024,
|
| 271 |
+
512,
|
| 272 |
+
1024,
|
| 273 |
+
1024,
|
| 274 |
+
640,
|
| 275 |
+
896,
|
| 276 |
+
640,
|
| 277 |
+
1024,
|
| 278 |
+
896,
|
| 279 |
+
384,
|
| 280 |
+
1024,
|
| 281 |
+
128,
|
| 282 |
+
896,
|
| 283 |
+
768,
|
| 284 |
+
1024,
|
| 285 |
+
768,
|
| 286 |
+
640,
|
| 287 |
+
896,
|
| 288 |
+
768,
|
| 289 |
+
640,
|
| 290 |
+
512,
|
| 291 |
+
896,
|
| 292 |
+
512,
|
| 293 |
+
640,
|
| 294 |
+
256,
|
| 295 |
+
768,
|
| 296 |
+
640,
|
| 297 |
+
768,
|
| 298 |
+
384,
|
| 299 |
+
896,
|
| 300 |
+
512,
|
| 301 |
+
512,
|
| 302 |
+
256,
|
| 303 |
+
512,
|
| 304 |
+
896,
|
| 305 |
+
256,
|
| 306 |
+
384,
|
| 307 |
+
640,
|
| 308 |
+
512,
|
| 309 |
+
640,
|
| 310 |
+
896,
|
| 311 |
+
512,
|
| 312 |
+
1024,
|
| 313 |
+
256,
|
| 314 |
+
768,
|
| 315 |
+
1024,
|
| 316 |
+
768,
|
| 317 |
+
256,
|
| 318 |
+
256
|
| 319 |
+
],
|
| 320 |
+
[
|
| 321 |
+
640,
|
| 322 |
+
896,
|
| 323 |
+
1024,
|
| 324 |
+
896,
|
| 325 |
+
1024,
|
| 326 |
+
1024,
|
| 327 |
+
1024,
|
| 328 |
+
512,
|
| 329 |
+
256,
|
| 330 |
+
256,
|
| 331 |
+
1024,
|
| 332 |
+
768,
|
| 333 |
+
512,
|
| 334 |
+
768,
|
| 335 |
+
1024,
|
| 336 |
+
1024,
|
| 337 |
+
1024,
|
| 338 |
+
384,
|
| 339 |
+
512,
|
| 340 |
+
1024,
|
| 341 |
+
512,
|
| 342 |
+
1024,
|
| 343 |
+
128,
|
| 344 |
+
640,
|
| 345 |
+
640,
|
| 346 |
+
896,
|
| 347 |
+
768,
|
| 348 |
+
128,
|
| 349 |
+
256,
|
| 350 |
+
256,
|
| 351 |
+
256,
|
| 352 |
+
256,
|
| 353 |
+
896,
|
| 354 |
+
1024,
|
| 355 |
+
1024,
|
| 356 |
+
384,
|
| 357 |
+
896,
|
| 358 |
+
256,
|
| 359 |
+
896,
|
| 360 |
+
640,
|
| 361 |
+
1024,
|
| 362 |
+
384,
|
| 363 |
+
640,
|
| 364 |
+
256,
|
| 365 |
+
1024,
|
| 366 |
+
1024,
|
| 367 |
+
1024
|
| 368 |
+
],
|
| 369 |
+
[
|
| 370 |
+
1024,
|
| 371 |
+
384,
|
| 372 |
+
1024,
|
| 373 |
+
1024,
|
| 374 |
+
256,
|
| 375 |
+
128,
|
| 376 |
+
256,
|
| 377 |
+
384,
|
| 378 |
+
256,
|
| 379 |
+
384,
|
| 380 |
+
896,
|
| 381 |
+
768,
|
| 382 |
+
896,
|
| 383 |
+
896,
|
| 384 |
+
512,
|
| 385 |
+
896,
|
| 386 |
+
640,
|
| 387 |
+
384,
|
| 388 |
+
384,
|
| 389 |
+
896,
|
| 390 |
+
768,
|
| 391 |
+
384,
|
| 392 |
+
896,
|
| 393 |
+
768,
|
| 394 |
+
768,
|
| 395 |
+
512,
|
| 396 |
+
896,
|
| 397 |
+
768,
|
| 398 |
+
768,
|
| 399 |
+
896,
|
| 400 |
+
128,
|
| 401 |
+
896,
|
| 402 |
+
512,
|
| 403 |
+
256,
|
| 404 |
+
768,
|
| 405 |
+
128,
|
| 406 |
+
384,
|
| 407 |
+
256,
|
| 408 |
+
896,
|
| 409 |
+
896,
|
| 410 |
+
384,
|
| 411 |
+
768,
|
| 412 |
+
512,
|
| 413 |
+
640,
|
| 414 |
+
256,
|
| 415 |
+
768,
|
| 416 |
+
640,
|
| 417 |
+
896,
|
| 418 |
+
384,
|
| 419 |
+
512,
|
| 420 |
+
1024,
|
| 421 |
+
768,
|
| 422 |
+
384
|
| 423 |
+
],
|
| 424 |
+
[
|
| 425 |
+
512,
|
| 426 |
+
768,
|
| 427 |
+
512,
|
| 428 |
+
256,
|
| 429 |
+
128,
|
| 430 |
+
640,
|
| 431 |
+
384,
|
| 432 |
+
640,
|
| 433 |
+
768,
|
| 434 |
+
896,
|
| 435 |
+
640,
|
| 436 |
+
768,
|
| 437 |
+
256,
|
| 438 |
+
384,
|
| 439 |
+
1024,
|
| 440 |
+
896,
|
| 441 |
+
256,
|
| 442 |
+
896,
|
| 443 |
+
512,
|
| 444 |
+
256,
|
| 445 |
+
896,
|
| 446 |
+
768,
|
| 447 |
+
256,
|
| 448 |
+
896,
|
| 449 |
+
896,
|
| 450 |
+
384,
|
| 451 |
+
896,
|
| 452 |
+
640,
|
| 453 |
+
768,
|
| 454 |
+
512,
|
| 455 |
+
768,
|
| 456 |
+
768,
|
| 457 |
+
1024,
|
| 458 |
+
768,
|
| 459 |
+
640,
|
| 460 |
+
768,
|
| 461 |
+
384,
|
| 462 |
+
256,
|
| 463 |
+
512,
|
| 464 |
+
896,
|
| 465 |
+
128,
|
| 466 |
+
384,
|
| 467 |
+
256,
|
| 468 |
+
768,
|
| 469 |
+
384,
|
| 470 |
+
256,
|
| 471 |
+
1024,
|
| 472 |
+
1024,
|
| 473 |
+
896,
|
| 474 |
+
256,
|
| 475 |
+
1024,
|
| 476 |
+
256,
|
| 477 |
+
128,
|
| 478 |
+
896
|
| 479 |
+
],
|
| 480 |
+
[
|
| 481 |
+
640,
|
| 482 |
+
640,
|
| 483 |
+
896,
|
| 484 |
+
256,
|
| 485 |
+
1024,
|
| 486 |
+
512,
|
| 487 |
+
1024,
|
| 488 |
+
768,
|
| 489 |
+
384,
|
| 490 |
+
512,
|
| 491 |
+
256,
|
| 492 |
+
768,
|
| 493 |
+
896,
|
| 494 |
+
768,
|
| 495 |
+
512,
|
| 496 |
+
768,
|
| 497 |
+
768,
|
| 498 |
+
640,
|
| 499 |
+
384,
|
| 500 |
+
768,
|
| 501 |
+
512,
|
| 502 |
+
768,
|
| 503 |
+
768,
|
| 504 |
+
512,
|
| 505 |
+
768,
|
| 506 |
+
128,
|
| 507 |
+
896,
|
| 508 |
+
512,
|
| 509 |
+
768,
|
| 510 |
+
1024,
|
| 511 |
+
128,
|
| 512 |
+
384,
|
| 513 |
+
768,
|
| 514 |
+
768,
|
| 515 |
+
768,
|
| 516 |
+
384,
|
| 517 |
+
512,
|
| 518 |
+
640,
|
| 519 |
+
768,
|
| 520 |
+
512,
|
| 521 |
+
768,
|
| 522 |
+
1024,
|
| 523 |
+
640,
|
| 524 |
+
896,
|
| 525 |
+
256,
|
| 526 |
+
1024,
|
| 527 |
+
384,
|
| 528 |
+
768,
|
| 529 |
+
768,
|
| 530 |
+
768
|
| 531 |
+
],
|
| 532 |
+
[
|
| 533 |
+
896,
|
| 534 |
+
512,
|
| 535 |
+
896,
|
| 536 |
+
768,
|
| 537 |
+
384,
|
| 538 |
+
384,
|
| 539 |
+
768,
|
| 540 |
+
512,
|
| 541 |
+
768,
|
| 542 |
+
512,
|
| 543 |
+
1024,
|
| 544 |
+
640,
|
| 545 |
+
896,
|
| 546 |
+
896,
|
| 547 |
+
256,
|
| 548 |
+
640,
|
| 549 |
+
1024,
|
| 550 |
+
256,
|
| 551 |
+
896,
|
| 552 |
+
128,
|
| 553 |
+
128,
|
| 554 |
+
128,
|
| 555 |
+
768,
|
| 556 |
+
896,
|
| 557 |
+
384,
|
| 558 |
+
896,
|
| 559 |
+
512,
|
| 560 |
+
896,
|
| 561 |
+
384,
|
| 562 |
+
256,
|
| 563 |
+
640,
|
| 564 |
+
640,
|
| 565 |
+
896,
|
| 566 |
+
768,
|
| 567 |
+
640,
|
| 568 |
+
256,
|
| 569 |
+
896,
|
| 570 |
+
896,
|
| 571 |
+
512,
|
| 572 |
+
128,
|
| 573 |
+
896,
|
| 574 |
+
256,
|
| 575 |
+
256,
|
| 576 |
+
640,
|
| 577 |
+
896,
|
| 578 |
+
896,
|
| 579 |
+
128,
|
| 580 |
+
1024,
|
| 581 |
+
256,
|
| 582 |
+
384,
|
| 583 |
+
1024,
|
| 584 |
+
640,
|
| 585 |
+
896
|
| 586 |
+
],
|
| 587 |
+
[
|
| 588 |
+
768,
|
| 589 |
+
384,
|
| 590 |
+
640,
|
| 591 |
+
896,
|
| 592 |
+
256,
|
| 593 |
+
128,
|
| 594 |
+
384,
|
| 595 |
+
896,
|
| 596 |
+
128,
|
| 597 |
+
128,
|
| 598 |
+
896,
|
| 599 |
+
256,
|
| 600 |
+
384,
|
| 601 |
+
896,
|
| 602 |
+
512,
|
| 603 |
+
768,
|
| 604 |
+
768,
|
| 605 |
+
512,
|
| 606 |
+
512,
|
| 607 |
+
768,
|
| 608 |
+
896,
|
| 609 |
+
640,
|
| 610 |
+
768,
|
| 611 |
+
896,
|
| 612 |
+
896,
|
| 613 |
+
896,
|
| 614 |
+
640,
|
| 615 |
+
896,
|
| 616 |
+
640,
|
| 617 |
+
512,
|
| 618 |
+
896,
|
| 619 |
+
256,
|
| 620 |
+
512,
|
| 621 |
+
128,
|
| 622 |
+
512,
|
| 623 |
+
384,
|
| 624 |
+
768,
|
| 625 |
+
768,
|
| 626 |
+
1024,
|
| 627 |
+
256,
|
| 628 |
+
768,
|
| 629 |
+
256,
|
| 630 |
+
768,
|
| 631 |
+
512,
|
| 632 |
+
640,
|
| 633 |
+
1024,
|
| 634 |
+
128,
|
| 635 |
+
896,
|
| 636 |
+
896,
|
| 637 |
+
896,
|
| 638 |
+
896,
|
| 639 |
+
1024
|
| 640 |
+
],
|
| 641 |
+
[
|
| 642 |
+
512,
|
| 643 |
+
384,
|
| 644 |
+
768,
|
| 645 |
+
640,
|
| 646 |
+
640,
|
| 647 |
+
768,
|
| 648 |
+
1024,
|
| 649 |
+
896,
|
| 650 |
+
512,
|
| 651 |
+
256,
|
| 652 |
+
640,
|
| 653 |
+
768,
|
| 654 |
+
640,
|
| 655 |
+
896,
|
| 656 |
+
128,
|
| 657 |
+
256,
|
| 658 |
+
896,
|
| 659 |
+
1024,
|
| 660 |
+
256,
|
| 661 |
+
640,
|
| 662 |
+
512,
|
| 663 |
+
256,
|
| 664 |
+
128,
|
| 665 |
+
512,
|
| 666 |
+
256,
|
| 667 |
+
640,
|
| 668 |
+
768,
|
| 669 |
+
768,
|
| 670 |
+
128,
|
| 671 |
+
128,
|
| 672 |
+
768,
|
| 673 |
+
640,
|
| 674 |
+
1024,
|
| 675 |
+
1024,
|
| 676 |
+
768,
|
| 677 |
+
512,
|
| 678 |
+
896,
|
| 679 |
+
768,
|
| 680 |
+
896,
|
| 681 |
+
1024,
|
| 682 |
+
896,
|
| 683 |
+
896,
|
| 684 |
+
512,
|
| 685 |
+
640,
|
| 686 |
+
1024,
|
| 687 |
+
512,
|
| 688 |
+
1024,
|
| 689 |
+
512,
|
| 690 |
+
512,
|
| 691 |
+
512,
|
| 692 |
+
768
|
| 693 |
+
],
|
| 694 |
+
[
|
| 695 |
+
896,
|
| 696 |
+
896,
|
| 697 |
+
1024,
|
| 698 |
+
1024,
|
| 699 |
+
896,
|
| 700 |
+
128,
|
| 701 |
+
768,
|
| 702 |
+
256,
|
| 703 |
+
1024,
|
| 704 |
+
256,
|
| 705 |
+
1024,
|
| 706 |
+
640,
|
| 707 |
+
384,
|
| 708 |
+
256,
|
| 709 |
+
256,
|
| 710 |
+
512,
|
| 711 |
+
768,
|
| 712 |
+
896,
|
| 713 |
+
512,
|
| 714 |
+
768,
|
| 715 |
+
384,
|
| 716 |
+
1024,
|
| 717 |
+
896,
|
| 718 |
+
896,
|
| 719 |
+
1024,
|
| 720 |
+
896,
|
| 721 |
+
768,
|
| 722 |
+
896,
|
| 723 |
+
640,
|
| 724 |
+
1024,
|
| 725 |
+
512,
|
| 726 |
+
896,
|
| 727 |
+
512,
|
| 728 |
+
1024,
|
| 729 |
+
512,
|
| 730 |
+
512,
|
| 731 |
+
256,
|
| 732 |
+
256,
|
| 733 |
+
256,
|
| 734 |
+
512,
|
| 735 |
+
768,
|
| 736 |
+
128,
|
| 737 |
+
384,
|
| 738 |
+
512,
|
| 739 |
+
896,
|
| 740 |
+
896,
|
| 741 |
+
1024,
|
| 742 |
+
256,
|
| 743 |
+
384,
|
| 744 |
+
640
|
| 745 |
+
],
|
| 746 |
+
[
|
| 747 |
+
896,
|
| 748 |
+
640,
|
| 749 |
+
384,
|
| 750 |
+
512,
|
| 751 |
+
256,
|
| 752 |
+
640,
|
| 753 |
+
1024,
|
| 754 |
+
384,
|
| 755 |
+
1024,
|
| 756 |
+
1024,
|
| 757 |
+
768,
|
| 758 |
+
256,
|
| 759 |
+
1024,
|
| 760 |
+
768,
|
| 761 |
+
512,
|
| 762 |
+
896,
|
| 763 |
+
256,
|
| 764 |
+
1024,
|
| 765 |
+
768,
|
| 766 |
+
768,
|
| 767 |
+
768,
|
| 768 |
+
384,
|
| 769 |
+
384,
|
| 770 |
+
256,
|
| 771 |
+
1024,
|
| 772 |
+
384,
|
| 773 |
+
384,
|
| 774 |
+
384,
|
| 775 |
+
896,
|
| 776 |
+
768,
|
| 777 |
+
640,
|
| 778 |
+
768,
|
| 779 |
+
512,
|
| 780 |
+
896,
|
| 781 |
+
896,
|
| 782 |
+
896,
|
| 783 |
+
896,
|
| 784 |
+
256,
|
| 785 |
+
384,
|
| 786 |
+
128,
|
| 787 |
+
1024,
|
| 788 |
+
896,
|
| 789 |
+
256,
|
| 790 |
+
256,
|
| 791 |
+
768,
|
| 792 |
+
640,
|
| 793 |
+
896,
|
| 794 |
+
384,
|
| 795 |
+
768,
|
| 796 |
+
512,
|
| 797 |
+
640
|
| 798 |
+
],
|
| 799 |
+
[
|
| 800 |
+
896,
|
| 801 |
+
1024,
|
| 802 |
+
768,
|
| 803 |
+
1024,
|
| 804 |
+
896,
|
| 805 |
+
256,
|
| 806 |
+
768,
|
| 807 |
+
128,
|
| 808 |
+
128,
|
| 809 |
+
768,
|
| 810 |
+
512,
|
| 811 |
+
896,
|
| 812 |
+
384,
|
| 813 |
+
768,
|
| 814 |
+
1024,
|
| 815 |
+
256,
|
| 816 |
+
768,
|
| 817 |
+
768,
|
| 818 |
+
256,
|
| 819 |
+
512,
|
| 820 |
+
512,
|
| 821 |
+
640,
|
| 822 |
+
512,
|
| 823 |
+
256,
|
| 824 |
+
768,
|
| 825 |
+
896,
|
| 826 |
+
384,
|
| 827 |
+
1024,
|
| 828 |
+
640,
|
| 829 |
+
1024,
|
| 830 |
+
512,
|
| 831 |
+
512,
|
| 832 |
+
384,
|
| 833 |
+
512,
|
| 834 |
+
512,
|
| 835 |
+
1024,
|
| 836 |
+
384,
|
| 837 |
+
896,
|
| 838 |
+
768,
|
| 839 |
+
384,
|
| 840 |
+
384,
|
| 841 |
+
128,
|
| 842 |
+
384,
|
| 843 |
+
1024,
|
| 844 |
+
896,
|
| 845 |
+
640,
|
| 846 |
+
768,
|
| 847 |
+
768,
|
| 848 |
+
256,
|
| 849 |
+
640,
|
| 850 |
+
512,
|
| 851 |
+
640,
|
| 852 |
+
384
|
| 853 |
+
]
|
| 854 |
+
],
|
| 855 |
+
"glean_metadata": {
|
| 856 |
+
"base_model": "allenai/OLMoE-1B-7B-0125-Instruct",
|
| 857 |
+
"block_size": 128,
|
| 858 |
+
"criterion": "reap",
|
| 859 |
+
"dead_experts": 217,
|
| 860 |
+
"keep_fraction": 0.5,
|
| 861 |
+
"min_width": 128,
|
| 862 |
+
"params": 3697491968,
|
| 863 |
+
"scores": "outputs/scores_0125inst_dolmino-math/scores.pt"
|
| 864 |
+
},
|
| 865 |
+
"hidden_act": "silu",
|
| 866 |
+
"hidden_size": 2048,
|
| 867 |
+
"initializer_range": 0.02,
|
| 868 |
+
"intermediate_size": 1024,
|
| 869 |
+
"max_position_embeddings": 4096,
|
| 870 |
+
"model_type": "pruned_olmoe",
|
| 871 |
+
"norm_topk_prob": false,
|
| 872 |
+
"num_attention_heads": 16,
|
| 873 |
+
"num_experts": 64,
|
| 874 |
+
"num_experts_per_tok": 8,
|
| 875 |
+
"num_hidden_layers": 16,
|
| 876 |
+
"num_key_value_heads": 16,
|
| 877 |
+
"output_router_logits": false,
|
| 878 |
+
"pad_token_id": 1,
|
| 879 |
+
"rms_norm_eps": 1e-05,
|
| 880 |
+
"rope_scaling": null,
|
| 881 |
+
"rope_theta": 10000.0,
|
| 882 |
+
"router_aux_loss_coef": 0.01,
|
| 883 |
+
"tie_word_embeddings": false,
|
| 884 |
+
"transformers_version": "4.57.6",
|
| 885 |
+
"use_cache": false,
|
| 886 |
+
"vocab_size": 50304
|
| 887 |
+
}
|
healed/opd_warm_unleashed/step0500/configuration_pruned_olmoe.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Configuration for GLEAN-pruned OLMoE: variable-width, variable-count experts.
|
| 2 |
+
"""
|
| 3 |
+
|
| 4 |
+
from transformers.models.olmoe.configuration_olmoe import OlmoeConfig
|
| 5 |
+
|
| 6 |
+
|
| 7 |
+
class PrunedOlmoeConfig(OlmoeConfig):
|
| 8 |
+
"""OlmoeConfig plus a per-(layer, expert) width table.
|
| 9 |
+
|
| 10 |
+
``expert_widths[l]`` lists the SwiGLU intermediate width of each surviving
|
| 11 |
+
expert in decoder layer ``l``, in expert order. Lists are ragged: layers
|
| 12 |
+
may keep different numbers of experts (deleted experts simply don't
|
| 13 |
+
appear — the router in layer ``l`` has ``len(expert_widths[l])`` rows),
|
| 14 |
+
and each width may differ (multiples of the GEMM block size, 128, for
|
| 15 |
+
variable-MegaBlocks execution). ``None`` means an unpruned model
|
| 16 |
+
(uniform ``num_experts`` × ``intermediate_size``).
|
| 17 |
+
|
| 18 |
+
The inherited ``num_experts`` / ``intermediate_size`` keep their ORIGINAL
|
| 19 |
+
(pre-pruning) values for provenance; the width table is authoritative for
|
| 20 |
+
the built architecture.
|
| 21 |
+
"""
|
| 22 |
+
|
| 23 |
+
model_type = "pruned_olmoe"
|
| 24 |
+
|
| 25 |
+
def __init__(self, expert_widths: list[list[int]] | None = None, **kwargs):
|
| 26 |
+
super().__init__(**kwargs)
|
| 27 |
+
self.expert_widths = expert_widths
|
healed/opd_warm_unleashed/step0500/generation_config.json
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_from_model_config": true,
|
| 3 |
+
"eos_token_id": 50279,
|
| 4 |
+
"pad_token_id": 1,
|
| 5 |
+
"transformers_version": "4.57.6"
|
| 6 |
+
}
|
healed/opd_warm_unleashed/step0500/model.safetensors.index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/opd_warm_unleashed/step0500/modeling_pruned_olmoe.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""GLEAN-pruned OLMoE: HF-loadable model with ragged (variable-width) experts.
|
| 2 |
+
|
| 3 |
+
Pattern follows hbfreed/variable-flex-olmo's PrunedFlexOlmoForCausalLM
|
| 4 |
+
(docs/recon/prior-work-hbfreed.md), generalized from one scalar width to a
|
| 5 |
+
per-(layer, expert) width table: ``super().__init__`` builds the uniform
|
| 6 |
+
architecture from the config, then every MoE block is rebuilt to its pruned
|
| 7 |
+
shape — surviving experts only, each at its own width, router sliced to
|
| 8 |
+
match — so the state dict aligns exactly with what
|
| 9 |
+
``glean.prune.prune_channels_global`` leaves behind.
|
| 10 |
+
|
| 11 |
+
Caveat: ``output_router_logits=True`` (the load-balancing aux loss) assumes a
|
| 12 |
+
uniform ``config.num_experts`` and is unsupported on ragged models.
|
| 13 |
+
"""
|
| 14 |
+
|
| 15 |
+
import torch.nn as nn
|
| 16 |
+
from transformers.activations import ACT2FN
|
| 17 |
+
from transformers.models.olmoe.modeling_olmoe import OlmoeForCausalLM
|
| 18 |
+
|
| 19 |
+
from .configuration_pruned_olmoe import PrunedOlmoeConfig
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
class RaggedOlmoeMLP(nn.Module):
|
| 23 |
+
"""OlmoeMLP with an explicit intermediate width (SwiGLU, no biases)."""
|
| 24 |
+
|
| 25 |
+
def __init__(self, hidden_size: int, intermediate_size: int, hidden_act: str):
|
| 26 |
+
super().__init__()
|
| 27 |
+
self.hidden_size = hidden_size
|
| 28 |
+
self.intermediate_size = intermediate_size
|
| 29 |
+
self.gate_proj = nn.Linear(hidden_size, intermediate_size, bias=False)
|
| 30 |
+
self.up_proj = nn.Linear(hidden_size, intermediate_size, bias=False)
|
| 31 |
+
self.down_proj = nn.Linear(intermediate_size, hidden_size, bias=False)
|
| 32 |
+
self.act_fn = ACT2FN[hidden_act]
|
| 33 |
+
|
| 34 |
+
def forward(self, x):
|
| 35 |
+
return self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 36 |
+
|
| 37 |
+
|
| 38 |
+
class PrunedOlmoeForCausalLM(OlmoeForCausalLM):
|
| 39 |
+
"""OLMoE with per-layer surviving-expert lists at per-expert widths."""
|
| 40 |
+
|
| 41 |
+
config_class = PrunedOlmoeConfig
|
| 42 |
+
|
| 43 |
+
def __init__(self, config: PrunedOlmoeConfig):
|
| 44 |
+
super().__init__(config)
|
| 45 |
+
widths_table = getattr(config, "expert_widths", None)
|
| 46 |
+
if widths_table is None:
|
| 47 |
+
return # unpruned: plain OLMoE
|
| 48 |
+
if len(widths_table) != len(self.model.layers):
|
| 49 |
+
raise ValueError(
|
| 50 |
+
f"expert_widths has {len(widths_table)} rows but the model has "
|
| 51 |
+
f"{len(self.model.layers)} decoder layers"
|
| 52 |
+
)
|
| 53 |
+
for layer, widths in zip(self.model.layers, widths_table):
|
| 54 |
+
if any(w <= 0 for w in widths):
|
| 55 |
+
raise ValueError("expert_widths must list surviving experts only (>0)")
|
| 56 |
+
block = layer.mlp
|
| 57 |
+
if len(widths) < block.top_k:
|
| 58 |
+
raise ValueError(
|
| 59 |
+
f"a layer keeps {len(widths)} experts < top_k={block.top_k}"
|
| 60 |
+
)
|
| 61 |
+
block.num_experts = len(widths)
|
| 62 |
+
block.gate = nn.Linear(config.hidden_size, len(widths), bias=False)
|
| 63 |
+
block.experts = nn.ModuleList(
|
| 64 |
+
RaggedOlmoeMLP(config.hidden_size, w, config.hidden_act)
|
| 65 |
+
for w in widths
|
| 66 |
+
)
|
healed/opd_warm_unleashed/step0500/special_tokens_map.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": {
|
| 3 |
+
"content": "|||IP_ADDRESS|||",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": true,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"eos_token": {
|
| 10 |
+
"content": "|||IP_ADDRESS|||",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": true,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"pad_token": {
|
| 17 |
+
"content": "<pad>",
|
| 18 |
+
"lstrip": false,
|
| 19 |
+
"normalized": false,
|
| 20 |
+
"rstrip": false,
|
| 21 |
+
"single_word": false
|
| 22 |
+
}
|
| 23 |
+
}
|
healed/opd_warm_unleashed/step0500/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/opd_warm_unleashed/step0500/tokenizer_config.json
ADDED
|
@@ -0,0 +1,247 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_bos_token": false,
|
| 3 |
+
"add_eos_token": false,
|
| 4 |
+
"add_prefix_space": false,
|
| 5 |
+
"added_tokens_decoder": {
|
| 6 |
+
"0": {
|
| 7 |
+
"content": "<|endoftext|>",
|
| 8 |
+
"lstrip": false,
|
| 9 |
+
"normalized": false,
|
| 10 |
+
"rstrip": false,
|
| 11 |
+
"single_word": false,
|
| 12 |
+
"special": true
|
| 13 |
+
},
|
| 14 |
+
"1": {
|
| 15 |
+
"content": "<|padding|>",
|
| 16 |
+
"lstrip": false,
|
| 17 |
+
"normalized": false,
|
| 18 |
+
"rstrip": false,
|
| 19 |
+
"single_word": false,
|
| 20 |
+
"special": true
|
| 21 |
+
},
|
| 22 |
+
"50254": {
|
| 23 |
+
"content": " ",
|
| 24 |
+
"lstrip": false,
|
| 25 |
+
"normalized": true,
|
| 26 |
+
"rstrip": false,
|
| 27 |
+
"single_word": false,
|
| 28 |
+
"special": false
|
| 29 |
+
},
|
| 30 |
+
"50255": {
|
| 31 |
+
"content": " ",
|
| 32 |
+
"lstrip": false,
|
| 33 |
+
"normalized": true,
|
| 34 |
+
"rstrip": false,
|
| 35 |
+
"single_word": false,
|
| 36 |
+
"special": false
|
| 37 |
+
},
|
| 38 |
+
"50256": {
|
| 39 |
+
"content": " ",
|
| 40 |
+
"lstrip": false,
|
| 41 |
+
"normalized": true,
|
| 42 |
+
"rstrip": false,
|
| 43 |
+
"single_word": false,
|
| 44 |
+
"special": false
|
| 45 |
+
},
|
| 46 |
+
"50257": {
|
| 47 |
+
"content": " ",
|
| 48 |
+
"lstrip": false,
|
| 49 |
+
"normalized": true,
|
| 50 |
+
"rstrip": false,
|
| 51 |
+
"single_word": false,
|
| 52 |
+
"special": false
|
| 53 |
+
},
|
| 54 |
+
"50258": {
|
| 55 |
+
"content": " ",
|
| 56 |
+
"lstrip": false,
|
| 57 |
+
"normalized": true,
|
| 58 |
+
"rstrip": false,
|
| 59 |
+
"single_word": false,
|
| 60 |
+
"special": false
|
| 61 |
+
},
|
| 62 |
+
"50259": {
|
| 63 |
+
"content": " ",
|
| 64 |
+
"lstrip": false,
|
| 65 |
+
"normalized": true,
|
| 66 |
+
"rstrip": false,
|
| 67 |
+
"single_word": false,
|
| 68 |
+
"special": false
|
| 69 |
+
},
|
| 70 |
+
"50260": {
|
| 71 |
+
"content": " ",
|
| 72 |
+
"lstrip": false,
|
| 73 |
+
"normalized": true,
|
| 74 |
+
"rstrip": false,
|
| 75 |
+
"single_word": false,
|
| 76 |
+
"special": false
|
| 77 |
+
},
|
| 78 |
+
"50261": {
|
| 79 |
+
"content": " ",
|
| 80 |
+
"lstrip": false,
|
| 81 |
+
"normalized": true,
|
| 82 |
+
"rstrip": false,
|
| 83 |
+
"single_word": false,
|
| 84 |
+
"special": false
|
| 85 |
+
},
|
| 86 |
+
"50262": {
|
| 87 |
+
"content": " ",
|
| 88 |
+
"lstrip": false,
|
| 89 |
+
"normalized": true,
|
| 90 |
+
"rstrip": false,
|
| 91 |
+
"single_word": false,
|
| 92 |
+
"special": false
|
| 93 |
+
},
|
| 94 |
+
"50263": {
|
| 95 |
+
"content": " ",
|
| 96 |
+
"lstrip": false,
|
| 97 |
+
"normalized": true,
|
| 98 |
+
"rstrip": false,
|
| 99 |
+
"single_word": false,
|
| 100 |
+
"special": false
|
| 101 |
+
},
|
| 102 |
+
"50264": {
|
| 103 |
+
"content": " ",
|
| 104 |
+
"lstrip": false,
|
| 105 |
+
"normalized": true,
|
| 106 |
+
"rstrip": false,
|
| 107 |
+
"single_word": false,
|
| 108 |
+
"special": false
|
| 109 |
+
},
|
| 110 |
+
"50265": {
|
| 111 |
+
"content": " ",
|
| 112 |
+
"lstrip": false,
|
| 113 |
+
"normalized": true,
|
| 114 |
+
"rstrip": false,
|
| 115 |
+
"single_word": false,
|
| 116 |
+
"special": false
|
| 117 |
+
},
|
| 118 |
+
"50266": {
|
| 119 |
+
"content": " ",
|
| 120 |
+
"lstrip": false,
|
| 121 |
+
"normalized": true,
|
| 122 |
+
"rstrip": false,
|
| 123 |
+
"single_word": false,
|
| 124 |
+
"special": false
|
| 125 |
+
},
|
| 126 |
+
"50267": {
|
| 127 |
+
"content": " ",
|
| 128 |
+
"lstrip": false,
|
| 129 |
+
"normalized": true,
|
| 130 |
+
"rstrip": false,
|
| 131 |
+
"single_word": false,
|
| 132 |
+
"special": false
|
| 133 |
+
},
|
| 134 |
+
"50268": {
|
| 135 |
+
"content": " ",
|
| 136 |
+
"lstrip": false,
|
| 137 |
+
"normalized": true,
|
| 138 |
+
"rstrip": false,
|
| 139 |
+
"single_word": false,
|
| 140 |
+
"special": false
|
| 141 |
+
},
|
| 142 |
+
"50269": {
|
| 143 |
+
"content": " ",
|
| 144 |
+
"lstrip": false,
|
| 145 |
+
"normalized": true,
|
| 146 |
+
"rstrip": false,
|
| 147 |
+
"single_word": false,
|
| 148 |
+
"special": false
|
| 149 |
+
},
|
| 150 |
+
"50270": {
|
| 151 |
+
"content": " ",
|
| 152 |
+
"lstrip": false,
|
| 153 |
+
"normalized": true,
|
| 154 |
+
"rstrip": false,
|
| 155 |
+
"single_word": false,
|
| 156 |
+
"special": false
|
| 157 |
+
},
|
| 158 |
+
"50271": {
|
| 159 |
+
"content": " ",
|
| 160 |
+
"lstrip": false,
|
| 161 |
+
"normalized": true,
|
| 162 |
+
"rstrip": false,
|
| 163 |
+
"single_word": false,
|
| 164 |
+
"special": false
|
| 165 |
+
},
|
| 166 |
+
"50272": {
|
| 167 |
+
"content": " ",
|
| 168 |
+
"lstrip": false,
|
| 169 |
+
"normalized": true,
|
| 170 |
+
"rstrip": false,
|
| 171 |
+
"single_word": false,
|
| 172 |
+
"special": false
|
| 173 |
+
},
|
| 174 |
+
"50273": {
|
| 175 |
+
"content": " ",
|
| 176 |
+
"lstrip": false,
|
| 177 |
+
"normalized": true,
|
| 178 |
+
"rstrip": false,
|
| 179 |
+
"single_word": false,
|
| 180 |
+
"special": false
|
| 181 |
+
},
|
| 182 |
+
"50274": {
|
| 183 |
+
"content": " ",
|
| 184 |
+
"lstrip": false,
|
| 185 |
+
"normalized": true,
|
| 186 |
+
"rstrip": false,
|
| 187 |
+
"single_word": false,
|
| 188 |
+
"special": false
|
| 189 |
+
},
|
| 190 |
+
"50275": {
|
| 191 |
+
"content": " ",
|
| 192 |
+
"lstrip": false,
|
| 193 |
+
"normalized": true,
|
| 194 |
+
"rstrip": false,
|
| 195 |
+
"single_word": false,
|
| 196 |
+
"special": false
|
| 197 |
+
},
|
| 198 |
+
"50276": {
|
| 199 |
+
"content": " ",
|
| 200 |
+
"lstrip": false,
|
| 201 |
+
"normalized": true,
|
| 202 |
+
"rstrip": false,
|
| 203 |
+
"single_word": false,
|
| 204 |
+
"special": false
|
| 205 |
+
},
|
| 206 |
+
"50277": {
|
| 207 |
+
"content": "|||EMAIL_ADDRESS|||",
|
| 208 |
+
"lstrip": false,
|
| 209 |
+
"normalized": true,
|
| 210 |
+
"rstrip": false,
|
| 211 |
+
"single_word": false,
|
| 212 |
+
"special": false
|
| 213 |
+
},
|
| 214 |
+
"50278": {
|
| 215 |
+
"content": "|||PHONE_NUMBER|||",
|
| 216 |
+
"lstrip": false,
|
| 217 |
+
"normalized": true,
|
| 218 |
+
"rstrip": false,
|
| 219 |
+
"single_word": false,
|
| 220 |
+
"special": false
|
| 221 |
+
},
|
| 222 |
+
"50279": {
|
| 223 |
+
"content": "|||IP_ADDRESS|||",
|
| 224 |
+
"lstrip": false,
|
| 225 |
+
"normalized": true,
|
| 226 |
+
"rstrip": false,
|
| 227 |
+
"single_word": false,
|
| 228 |
+
"special": true
|
| 229 |
+
},
|
| 230 |
+
"50280": {
|
| 231 |
+
"content": "<pad>",
|
| 232 |
+
"lstrip": false,
|
| 233 |
+
"normalized": false,
|
| 234 |
+
"rstrip": false,
|
| 235 |
+
"single_word": false,
|
| 236 |
+
"special": true
|
| 237 |
+
}
|
| 238 |
+
},
|
| 239 |
+
"bos_token": "|||IP_ADDRESS|||",
|
| 240 |
+
"clean_up_tokenization_spaces": false,
|
| 241 |
+
"eos_token": "|||IP_ADDRESS|||",
|
| 242 |
+
"extra_special_tokens": {},
|
| 243 |
+
"model_max_length": 1000000000000000019884624838656,
|
| 244 |
+
"pad_token": "<pad>",
|
| 245 |
+
"tokenizer_class": "GPTNeoXTokenizer",
|
| 246 |
+
"unk_token": null
|
| 247 |
+
}
|
healed/opd_warm_unleashed/train_log.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/opd_warm_unleashed/vllm_server.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
healed/opd_warm_unleashed/wandb/debug-internal.log
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2026-08-02T20:32:18.760787811-07:00","level":"INFO","msg":"wandb-core"}
|
| 2 |
+
{"time":"2026-08-02T20:32:18.760821712-07:00","level":"INFO","msg":"stream: starting","core version":"0.28.0"}
|
| 3 |
+
{"time":"2026-08-02T20:32:18.874378042-07:00","level":"WARN","msg":"featurechecker: GraphQL client is nil, skipping feature loading"}
|
| 4 |
+
{"time":"2026-08-02T20:32:18.874396933-07:00","level":"WARN","msg":"featurechecker: GraphQL client is nil, skipping feature loading"}
|
| 5 |
+
{"time":"2026-08-02T20:32:18.874419894-07:00","level":"INFO","msg":"stream: created new stream","id":"j33a2nw2"}
|
| 6 |
+
{"time":"2026-08-02T20:32:18.874538027-07:00","level":"INFO","msg":"stream: started"}
|
| 7 |
+
{"time":"2026-08-02T20:32:18.874577258-07:00","level":"INFO","msg":"writer: started","stream_id":"j33a2nw2"}
|
| 8 |
+
{"time":"2026-08-02T20:32:18.87461865-07:00","level":"INFO","msg":"sender: started"}
|
| 9 |
+
{"time":"2026-08-02T20:32:18.874662091-07:00","level":"INFO","msg":"handler: started"}
|
| 10 |
+
{"time":"2026-08-02T20:32:18.891498801-07:00","level":"WARN","msg":"featurechecker: GraphQL client is nil, skipping feature loading"}
|
| 11 |
+
{"time":"2026-08-02T20:32:18.891525461-07:00","level":"WARN","msg":"runupserter: server does not expand metric globs but the x_server_side_expand_glob_metrics setting is set; ignoring"}
|
| 12 |
+
{"time":"2026-08-03T03:53:25.789200783-07:00","level":"INFO","msg":"handler: operation stats","stats":{}}
|
| 13 |
+
{"time":"2026-08-03T03:53:25.792541472-07:00","level":"INFO","msg":"stream: finishing up"}
|
| 14 |
+
{"time":"2026-08-03T03:53:25.792552172-07:00","level":"INFO","msg":"handler: closed"}
|
| 15 |
+
{"time":"2026-08-03T03:53:25.792600184-07:00","level":"INFO","msg":"sender: closed"}
|
| 16 |
+
{"time":"2026-08-03T03:53:25.792607014-07:00","level":"INFO","msg":"stream: all finished"}
|