Add files using upload-large-folder tool
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- evals/general_suite/healed/glean_keep25_s1224/lm_eval.log +0 -0
- evals/general_suite/healed/glean_keep25_s1224/lm_eval_code.log +27 -0
- evals/general_suite/healed/glean_keep25_s1224/server.log +0 -0
- evals/general_suite/healed/glean_keep25_s1224/student/results_2026-07-17T14-56-42.779609.json +337 -0
- evals/general_suite/healed/glean_keep25_s1224/student/results_2026-07-17T15-01-00.225834.json +225 -0
- evals/general_suite/healed/glean_keep25_s1224/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-56-42.779609.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1224/student/samples_humaneval_2026-07-17T15-01-00.225834.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1224/student/samples_ifeval_2026-07-17T14-56-42.779609.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1224/student/samples_mbpp_2026-07-17T15-01-00.225834.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1224/student/samples_minerva_math500_2026-07-17T14-56-42.779609.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1225/student/results_2026-07-17T14-48-07.537542.json +337 -0
- evals/general_suite/healed/glean_keep25_s1225/student/results_2026-07-17T14-54-29.755565.json +225 -0
- evals/general_suite/healed/glean_keep25_s1225/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-48-07.537542.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1225/student/samples_humaneval_2026-07-17T14-54-29.755565.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1225/student/samples_ifeval_2026-07-17T14-48-07.537542.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1225/student/samples_mbpp_2026-07-17T14-54-29.755565.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1225/student/samples_minerva_math500_2026-07-17T14-48-07.537542.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1226/lm_eval.log +0 -0
- evals/general_suite/healed/glean_keep25_s1226/lm_eval_code.log +27 -0
- evals/general_suite/healed/glean_keep25_s1226/server.log +0 -0
- evals/general_suite/healed/glean_keep25_s1226/student/results_2026-07-17T14-48-34.159099.json +337 -0
- evals/general_suite/healed/glean_keep25_s1226/student/results_2026-07-17T14-54-44.918246.json +225 -0
- evals/general_suite/healed/glean_keep25_s1226/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-48-34.159099.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1226/student/samples_humaneval_2026-07-17T14-54-44.918246.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1226/student/samples_ifeval_2026-07-17T14-48-34.159099.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1226/student/samples_mbpp_2026-07-17T14-54-44.918246.jsonl +0 -0
- evals/general_suite/healed/glean_keep25_s1226/student/samples_minerva_math500_2026-07-17T14-48-34.159099.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224/student/results_2026-07-18T10-11-54.989656.json +337 -0
- evals/general_suite/healed/glean_keep50_s1224/student/results_2026-07-18T10-15-21.380553.json +225 -0
- evals/general_suite/healed/glean_keep50_s1224/student/samples_gsm8k_cot_zeroshot_2026-07-18T10-11-54.989656.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224/student/samples_humaneval_2026-07-18T10-15-21.380553.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224/student/samples_ifeval_2026-07-18T10-11-54.989656.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224/student/samples_mbpp_2026-07-18T10-15-21.380553.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224/student/samples_minerva_math500_2026-07-18T10-11-54.989656.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/results_2026-07-19T21-36-45.489848.json +337 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/results_2026-07-19T21-40-17.349569.json +225 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_gsm8k_cot_zeroshot_2026-07-19T21-36-45.489848.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_humaneval_2026-07-19T21-40-17.349569.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_ifeval_2026-07-19T21-36-45.489848.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_mbpp_2026-07-19T21-40-17.349569.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_minerva_math500_2026-07-19T21-36-45.489848.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/results_2026-07-19T21-49-11.345128.json +337 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/results_2026-07-19T21-52-17.977663.json +225 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_gsm8k_cot_zeroshot_2026-07-19T21-49-11.345128.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_humaneval_2026-07-19T21-52-17.977663.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_ifeval_2026-07-19T21-49-11.345128.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_mbpp_2026-07-19T21-52-17.977663.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_minerva_math500_2026-07-19T21-49-11.345128.jsonl +0 -0
- evals/general_suite/healed/glean_keep50_s1224_step100/student/results_2026-07-18T10-26-00.525964.json +337 -0
- evals/general_suite/healed/glean_keep50_s1224_step100/student/results_2026-07-18T10-29-19.004514.json +225 -0
evals/general_suite/healed/glean_keep25_s1224/lm_eval.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1224/lm_eval_code.log
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/164 [00:00<?, ?it/s]
|
| 1 |
89%|████████▉ | 146/164 [00:00<00:00, 1454.83it/s]
|
|
|
|
|
|
|
| 2 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 3 |
3%|▎ | 17/500 [00:00<00:02, 162.50it/s]
|
| 4 |
7%|▋ | 34/500 [00:00<00:02, 162.43it/s]
|
| 5 |
10%|█ | 51/500 [00:00<00:02, 163.22it/s]
|
| 6 |
14%|█▎ | 68/500 [00:00<00:02, 163.57it/s]
|
| 7 |
17%|█▋ | 85/500 [00:00<00:02, 164.04it/s]
|
| 8 |
20%|██ | 102/500 [00:00<00:02, 164.63it/s]
|
| 9 |
24%|██▍ | 119/500 [00:00<00:02, 165.03it/s]
|
| 10 |
27%|██▋ | 136/500 [00:00<00:02, 164.92it/s]
|
| 11 |
31%|███ | 153/500 [00:00<00:02, 164.85it/s]
|
| 12 |
34%|███▍ | 170/500 [00:01<00:01, 165.15it/s]
|
| 13 |
37%|███▋ | 187/500 [00:01<00:01, 165.38it/s]
|
| 14 |
41%|████ | 204/500 [00:01<00:01, 165.09it/s]
|
| 15 |
44%|████▍ | 221/500 [00:01<00:01, 165.27it/s]
|
| 16 |
48%|████▊ | 238/500 [00:01<00:01, 165.35it/s]
|
| 17 |
51%|█████ | 255/500 [00:01<00:01, 165.52it/s]
|
| 18 |
54%|█████▍ | 272/500 [00:01<00:01, 165.75it/s]
|
| 19 |
58%|█████▊ | 289/500 [00:01<00:01, 165.94it/s]
|
| 20 |
61%|██████ | 306/500 [00:01<00:01, 165.69it/s]
|
| 21 |
65%|██████▍ | 323/500 [00:01<00:01, 165.94it/s]
|
| 22 |
68%|██████▊ | 340/500 [00:02<00:00, 166.00it/s]
|
| 23 |
71%|███████▏ | 357/500 [00:02<00:00, 166.04it/s]
|
| 24 |
75%|███████▍ | 374/500 [00:02<00:00, 165.69it/s]
|
| 25 |
78%|███████▊ | 391/500 [00:02<00:00, 165.72it/s]
|
| 26 |
82%|████████▏ | 408/500 [00:02<00:00, 165.76it/s]
|
| 27 |
85%|████████▌ | 425/500 [00:02<00:00, 165.87it/s]
|
| 28 |
88%|████████▊ | 442/500 [00:02<00:00, 166.13it/s]
|
| 29 |
92%|█████████▏| 459/500 [00:02<00:00, 166.11it/s]
|
| 30 |
95%|█████████▌| 476/500 [00:02<00:00, 165.83it/s]
|
| 31 |
99%|█████████▊| 493/500 [00:02<00:00, 165.97it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17:14:56:52 INFO [_cli.run:388] Selected Tasks: ['humaneval', 'mbpp']
|
| 2 |
+
2026-07-17:14:56:53 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 3 |
+
2026-07-17:14:56:53 INFO [evaluator:239] Initializing local-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8420/v1/completions', 'tokenizer': 'outputs/healed/grid_general/glean_keep25_s1224/step0150', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 4 |
+
2026-07-17:14:56:53 INFO [models.openai_completions:42] Remote tokenizer not supported. Using huggingface tokenizer backend.
|
| 5 |
+
2026-07-17:14:56:53 INFO [models.api_models:179] Using max length 2048 - 1
|
| 6 |
+
2026-07-17:14:56:53 INFO [models.api_models:200] Using tokenizer huggingface
|
| 7 |
+
2026-07-17:14:56:59 INFO [evaluator_utils:446] Selected tasks:
|
| 8 |
+
2026-07-17:14:56:59 INFO [evaluator_utils:480] Task: humaneval (humaneval/humaneval.yaml)
|
| 9 |
+
2026-07-17:14:56:59 INFO [evaluator_utils:480] Task: mbpp (mbpp/mbpp.yaml)
|
| 10 |
+
2026-07-17:14:56:59 INFO [evaluator:314] humaneval: Using gen_kwargs: {'until': ['\nclass', '\ndef', '\n#', '\nif', '\nprint'], 'max_gen_toks': 1024, 'do_sample': False}
|
| 11 |
+
2026-07-17:14:56:59 INFO [evaluator:314] mbpp: Using gen_kwargs: {'until': ['[DONE]'], 'do_sample': False}
|
| 12 |
+
2026-07-17:14:56:59 INFO [api.task:312] Building contexts for humaneval on rank 0...
|
| 13 |
+
|
| 14 |
0%| | 0/164 [00:00<?, ?it/s]
|
| 15 |
89%|████████▉ | 146/164 [00:00<00:00, 1454.83it/s]
|
| 16 |
+
2026-07-17:14:57:00 INFO [api.task:312] Building contexts for mbpp on rank 0...
|
| 17 |
+
|
| 18 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 19 |
3%|▎ | 17/500 [00:00<00:02, 162.50it/s]
|
| 20 |
7%|▋ | 34/500 [00:00<00:02, 162.43it/s]
|
| 21 |
10%|█ | 51/500 [00:00<00:02, 163.22it/s]
|
| 22 |
14%|█▎ | 68/500 [00:00<00:02, 163.57it/s]
|
| 23 |
17%|█▋ | 85/500 [00:00<00:02, 164.04it/s]
|
| 24 |
20%|██ | 102/500 [00:00<00:02, 164.63it/s]
|
| 25 |
24%|██▍ | 119/500 [00:00<00:02, 165.03it/s]
|
| 26 |
27%|██▋ | 136/500 [00:00<00:02, 164.92it/s]
|
| 27 |
31%|███ | 153/500 [00:00<00:02, 164.85it/s]
|
| 28 |
34%|███▍ | 170/500 [00:01<00:01, 165.15it/s]
|
| 29 |
37%|███▋ | 187/500 [00:01<00:01, 165.38it/s]
|
| 30 |
41%|████ | 204/500 [00:01<00:01, 165.09it/s]
|
| 31 |
44%|████▍ | 221/500 [00:01<00:01, 165.27it/s]
|
| 32 |
48%|████▊ | 238/500 [00:01<00:01, 165.35it/s]
|
| 33 |
51%|█████ | 255/500 [00:01<00:01, 165.52it/s]
|
| 34 |
54%|█████▍ | 272/500 [00:01<00:01, 165.75it/s]
|
| 35 |
58%|█████▊ | 289/500 [00:01<00:01, 165.94it/s]
|
| 36 |
61%|██████ | 306/500 [00:01<00:01, 165.69it/s]
|
| 37 |
65%|██████▍ | 323/500 [00:01<00:01, 165.94it/s]
|
| 38 |
68%|██████▊ | 340/500 [00:02<00:00, 166.00it/s]
|
| 39 |
71%|███████▏ | 357/500 [00:02<00:00, 166.04it/s]
|
| 40 |
75%|███████▍ | 374/500 [00:02<00:00, 165.69it/s]
|
| 41 |
78%|███████▊ | 391/500 [00:02<00:00, 165.72it/s]
|
| 42 |
82%|████████▏ | 408/500 [00:02<00:00, 165.76it/s]
|
| 43 |
85%|████████▌ | 425/500 [00:02<00:00, 165.87it/s]
|
| 44 |
88%|████████▊ | 442/500 [00:02<00:00, 166.13it/s]
|
| 45 |
92%|█████████▏| 459/500 [00:02<00:00, 166.11it/s]
|
| 46 |
95%|█████████▌| 476/500 [00:02<00:00, 165.83it/s]
|
| 47 |
99%|█████████▊| 493/500 [00:02<00:00, 165.97it/s]
|
| 48 |
+
2026-07-17:14:57:03 INFO [evaluator:585] Running generate_until requests
|
| 49 |
+
2026-07-17:14:57:03 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
2026-07-17:15:01:00 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 53 |
+
2026-07-17:15:01:00 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/healed/glean_keep25_s1224/student/*.jsonl
|
| 54 |
+
local-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8420/v1/completions', 'tokenizer': 'outputs/healed/grid_general/glean_keep25_s1224/step0150', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: None, num_fewshot: None, batch_size: 1
|
| 55 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value | |Stderr|
|
| 56 |
+
|---------|------:|-----------|-----:|---------|---|-----:|---|-----:|
|
| 57 |
+
|humaneval| 1|create_test| 0|pass@1 |↑ |0.0183|± |0.0105|
|
| 58 |
+
|mbpp | 1|none | 3|pass_at_1|↑ |0.0940|± |0.0131|
|
| 59 |
+
|
evals/general_suite/healed/glean_keep25_s1224/server.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1224/student/results_2026-07-17T14-56-42.779609.json
ADDED
|
@@ -0,0 +1,337 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"gsm8k_cot_zeroshot": {
|
| 4 |
+
"name": "gsm8k_cot_zeroshot",
|
| 5 |
+
"alias": "gsm8k_cot_zeroshot",
|
| 6 |
+
"sample_len": 1319,
|
| 7 |
+
"exact_match,strict-match": 0.000758150113722517,
|
| 8 |
+
"exact_match_stderr,strict-match": 0.0007581501137225401,
|
| 9 |
+
"exact_match,flexible-extract": 0.19484457922668688,
|
| 10 |
+
"exact_match_stderr,flexible-extract": 0.01091003940957876
|
| 11 |
+
},
|
| 12 |
+
"minerva_math500": {
|
| 13 |
+
"name": "minerva_math500",
|
| 14 |
+
"alias": "minerva_math500",
|
| 15 |
+
"sample_len": 500,
|
| 16 |
+
"exact_match,none": 0.02,
|
| 17 |
+
"exact_match_stderr,none": 0.006267260734501833,
|
| 18 |
+
"math_verify,none": 0.08,
|
| 19 |
+
"math_verify_stderr,none": 0.012144751540478706
|
| 20 |
+
},
|
| 21 |
+
"ifeval": {
|
| 22 |
+
"name": "ifeval",
|
| 23 |
+
"alias": "ifeval",
|
| 24 |
+
"sample_len": 541,
|
| 25 |
+
"prompt_level_strict_acc,none": 0.4343807763401109,
|
| 26 |
+
"prompt_level_strict_acc_stderr,none": 0.02133047365756471,
|
| 27 |
+
"inst_level_strict_acc,none": 0.5635491606714629,
|
| 28 |
+
"inst_level_strict_acc_stderr,none": "N/A",
|
| 29 |
+
"prompt_level_loose_acc,none": 0.4584103512014787,
|
| 30 |
+
"prompt_level_loose_acc_stderr,none": 0.021442010560476468,
|
| 31 |
+
"inst_level_loose_acc,none": 0.5875299760191847,
|
| 32 |
+
"inst_level_loose_acc_stderr,none": "N/A"
|
| 33 |
+
}
|
| 34 |
+
},
|
| 35 |
+
"group_subtasks": {},
|
| 36 |
+
"configs": {
|
| 37 |
+
"gsm8k_cot_zeroshot": {
|
| 38 |
+
"task": "gsm8k_cot_zeroshot",
|
| 39 |
+
"dataset_path": "openai/gsm8k",
|
| 40 |
+
"dataset_name": "main",
|
| 41 |
+
"training_split": "train",
|
| 42 |
+
"test_split": "test",
|
| 43 |
+
"fewshot_split": "train",
|
| 44 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 45 |
+
"doc_to_target": "{{answer}}",
|
| 46 |
+
"unsafe_code": false,
|
| 47 |
+
"description": "",
|
| 48 |
+
"target_delimiter": " ",
|
| 49 |
+
"fewshot_delimiter": "\n\n",
|
| 50 |
+
"fewshot_config": {
|
| 51 |
+
"sampler": "default",
|
| 52 |
+
"split": "train",
|
| 53 |
+
"process_docs": null,
|
| 54 |
+
"fewshot_indices": null,
|
| 55 |
+
"samples": null,
|
| 56 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 57 |
+
"doc_to_choice": null,
|
| 58 |
+
"doc_to_target": "{{answer}}",
|
| 59 |
+
"gen_prefix": null,
|
| 60 |
+
"fewshot_delimiter": "\n\n",
|
| 61 |
+
"target_delimiter": " "
|
| 62 |
+
},
|
| 63 |
+
"num_fewshot": 0,
|
| 64 |
+
"metric_list": [
|
| 65 |
+
{
|
| 66 |
+
"metric": "exact_match",
|
| 67 |
+
"aggregation": "mean",
|
| 68 |
+
"higher_is_better": true,
|
| 69 |
+
"ignore_case": true,
|
| 70 |
+
"ignore_punctuation": false,
|
| 71 |
+
"regexes_to_ignore": [
|
| 72 |
+
",",
|
| 73 |
+
"\\$",
|
| 74 |
+
"(?s).*#### ",
|
| 75 |
+
"\\.$"
|
| 76 |
+
]
|
| 77 |
+
}
|
| 78 |
+
],
|
| 79 |
+
"output_type": "generate_until",
|
| 80 |
+
"generation_kwargs": {
|
| 81 |
+
"until": [
|
| 82 |
+
"Q:",
|
| 83 |
+
"</s>",
|
| 84 |
+
"<|im_end|>"
|
| 85 |
+
],
|
| 86 |
+
"do_sample": false,
|
| 87 |
+
"max_gen_toks": 1280
|
| 88 |
+
},
|
| 89 |
+
"repeats": 1,
|
| 90 |
+
"filter_list": [
|
| 91 |
+
{
|
| 92 |
+
"name": "strict-match",
|
| 93 |
+
"filter": [
|
| 94 |
+
{
|
| 95 |
+
"function": "regex",
|
| 96 |
+
"regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"function": "take_first"
|
| 100 |
+
}
|
| 101 |
+
]
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"name": "flexible-extract",
|
| 105 |
+
"filter": [
|
| 106 |
+
{
|
| 107 |
+
"function": "regex",
|
| 108 |
+
"group_select": -1,
|
| 109 |
+
"regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"function": "take_first"
|
| 113 |
+
}
|
| 114 |
+
]
|
| 115 |
+
}
|
| 116 |
+
],
|
| 117 |
+
"should_decontaminate": false,
|
| 118 |
+
"metadata": {
|
| 119 |
+
"version": 3.0,
|
| 120 |
+
"model": "student",
|
| 121 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 122 |
+
"num_concurrent": 48,
|
| 123 |
+
"tokenized_requests": false,
|
| 124 |
+
"max_retries": 3,
|
| 125 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
|
| 126 |
+
}
|
| 127 |
+
},
|
| 128 |
+
"ifeval": {
|
| 129 |
+
"task": "ifeval",
|
| 130 |
+
"dataset_path": "google/IFEval",
|
| 131 |
+
"test_split": "train",
|
| 132 |
+
"doc_to_text": "prompt",
|
| 133 |
+
"doc_to_target": 0,
|
| 134 |
+
"unsafe_code": false,
|
| 135 |
+
"process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
|
| 136 |
+
"description": "",
|
| 137 |
+
"target_delimiter": " ",
|
| 138 |
+
"fewshot_delimiter": "\n\n",
|
| 139 |
+
"fewshot_config": {
|
| 140 |
+
"sampler": "default",
|
| 141 |
+
"split": null,
|
| 142 |
+
"process_docs": null,
|
| 143 |
+
"fewshot_indices": null,
|
| 144 |
+
"samples": null,
|
| 145 |
+
"doc_to_text": "prompt",
|
| 146 |
+
"doc_to_choice": null,
|
| 147 |
+
"doc_to_target": 0,
|
| 148 |
+
"gen_prefix": null,
|
| 149 |
+
"fewshot_delimiter": "\n\n",
|
| 150 |
+
"target_delimiter": " "
|
| 151 |
+
},
|
| 152 |
+
"num_fewshot": 0,
|
| 153 |
+
"metric_list": [
|
| 154 |
+
{
|
| 155 |
+
"metric": "prompt_level_strict_acc",
|
| 156 |
+
"aggregation": "mean",
|
| 157 |
+
"higher_is_better": true
|
| 158 |
+
},
|
| 159 |
+
{
|
| 160 |
+
"metric": "inst_level_strict_acc",
|
| 161 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 162 |
+
"higher_is_better": true
|
| 163 |
+
},
|
| 164 |
+
{
|
| 165 |
+
"metric": "prompt_level_loose_acc",
|
| 166 |
+
"aggregation": "mean",
|
| 167 |
+
"higher_is_better": true
|
| 168 |
+
},
|
| 169 |
+
{
|
| 170 |
+
"metric": "inst_level_loose_acc",
|
| 171 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 172 |
+
"higher_is_better": true
|
| 173 |
+
}
|
| 174 |
+
],
|
| 175 |
+
"output_type": "generate_until",
|
| 176 |
+
"generation_kwargs": {
|
| 177 |
+
"until": [],
|
| 178 |
+
"do_sample": false,
|
| 179 |
+
"temperature": 0.0,
|
| 180 |
+
"max_gen_toks": 1280
|
| 181 |
+
},
|
| 182 |
+
"repeats": 1,
|
| 183 |
+
"should_decontaminate": false,
|
| 184 |
+
"metadata": {
|
| 185 |
+
"version": 4.0,
|
| 186 |
+
"model": "student",
|
| 187 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 188 |
+
"num_concurrent": 48,
|
| 189 |
+
"tokenized_requests": false,
|
| 190 |
+
"max_retries": 3,
|
| 191 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
|
| 192 |
+
}
|
| 193 |
+
},
|
| 194 |
+
"minerva_math500": {
|
| 195 |
+
"task": "minerva_math500",
|
| 196 |
+
"dataset_path": "HuggingFaceH4/MATH-500",
|
| 197 |
+
"dataset_name": "default",
|
| 198 |
+
"test_split": "test",
|
| 199 |
+
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
|
| 200 |
+
"doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
|
| 201 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 202 |
+
"unsafe_code": false,
|
| 203 |
+
"process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
|
| 204 |
+
"description": "",
|
| 205 |
+
"target_delimiter": " ",
|
| 206 |
+
"fewshot_delimiter": "\n\n",
|
| 207 |
+
"fewshot_config": {
|
| 208 |
+
"sampler": "first_n",
|
| 209 |
+
"split": null,
|
| 210 |
+
"process_docs": "<function process_docs at 0x7df0cb235b20>",
|
| 211 |
+
"fewshot_indices": null,
|
| 212 |
+
"samples": "<function list_fewshot_samples at 0x7df0cb2379c0>",
|
| 213 |
+
"doc_to_text": "<function doc_to_text at 0x7df0d0801800>",
|
| 214 |
+
"doc_to_choice": null,
|
| 215 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 216 |
+
"gen_prefix": null,
|
| 217 |
+
"fewshot_delimiter": "\n\n",
|
| 218 |
+
"target_delimiter": " "
|
| 219 |
+
},
|
| 220 |
+
"num_fewshot": 4,
|
| 221 |
+
"metric_list": [
|
| 222 |
+
{
|
| 223 |
+
"metric": "exact_match",
|
| 224 |
+
"aggregation": "mean",
|
| 225 |
+
"higher_is_better": true
|
| 226 |
+
},
|
| 227 |
+
{
|
| 228 |
+
"metric": "math_verify",
|
| 229 |
+
"aggregation": "mean",
|
| 230 |
+
"higher_is_better": true
|
| 231 |
+
}
|
| 232 |
+
],
|
| 233 |
+
"output_type": "generate_until",
|
| 234 |
+
"generation_kwargs": {
|
| 235 |
+
"until": [
|
| 236 |
+
"Problem:"
|
| 237 |
+
],
|
| 238 |
+
"do_sample": false,
|
| 239 |
+
"temperature": 0.0,
|
| 240 |
+
"max_gen_toks": 1280
|
| 241 |
+
},
|
| 242 |
+
"repeats": 1,
|
| 243 |
+
"should_decontaminate": false,
|
| 244 |
+
"metadata": {
|
| 245 |
+
"version": 3.0,
|
| 246 |
+
"model": "student",
|
| 247 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 248 |
+
"num_concurrent": 48,
|
| 249 |
+
"tokenized_requests": false,
|
| 250 |
+
"max_retries": 3,
|
| 251 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
|
| 252 |
+
}
|
| 253 |
+
}
|
| 254 |
+
},
|
| 255 |
+
"versions": {
|
| 256 |
+
"gsm8k_cot_zeroshot": 3.0,
|
| 257 |
+
"ifeval": 4.0,
|
| 258 |
+
"minerva_math500": 3.0
|
| 259 |
+
},
|
| 260 |
+
"n-shot": {
|
| 261 |
+
"gsm8k_cot_zeroshot": 0,
|
| 262 |
+
"ifeval": 0,
|
| 263 |
+
"minerva_math500": 4
|
| 264 |
+
},
|
| 265 |
+
"higher_is_better": {
|
| 266 |
+
"gsm8k_cot_zeroshot": {
|
| 267 |
+
"exact_match": true
|
| 268 |
+
},
|
| 269 |
+
"ifeval": {
|
| 270 |
+
"prompt_level_strict_acc": true,
|
| 271 |
+
"inst_level_strict_acc": true,
|
| 272 |
+
"prompt_level_loose_acc": true,
|
| 273 |
+
"inst_level_loose_acc": true
|
| 274 |
+
},
|
| 275 |
+
"minerva_math500": {
|
| 276 |
+
"exact_match": true,
|
| 277 |
+
"math_verify": true
|
| 278 |
+
}
|
| 279 |
+
},
|
| 280 |
+
"n-samples": {
|
| 281 |
+
"gsm8k_cot_zeroshot": {
|
| 282 |
+
"original": 1319,
|
| 283 |
+
"effective": 1319
|
| 284 |
+
},
|
| 285 |
+
"minerva_math500": {
|
| 286 |
+
"original": 500,
|
| 287 |
+
"effective": 500
|
| 288 |
+
},
|
| 289 |
+
"ifeval": {
|
| 290 |
+
"original": 541,
|
| 291 |
+
"effective": 541
|
| 292 |
+
}
|
| 293 |
+
},
|
| 294 |
+
"config": {
|
| 295 |
+
"model": "local-chat-completions",
|
| 296 |
+
"model_args": {
|
| 297 |
+
"model": "student",
|
| 298 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 299 |
+
"num_concurrent": 48,
|
| 300 |
+
"tokenized_requests": false,
|
| 301 |
+
"max_retries": 3
|
| 302 |
+
},
|
| 303 |
+
"batch_size": 1,
|
| 304 |
+
"batch_sizes": [],
|
| 305 |
+
"device": "cuda:0",
|
| 306 |
+
"use_cache": null,
|
| 307 |
+
"limit": null,
|
| 308 |
+
"bootstrap_iters": 100000,
|
| 309 |
+
"gen_kwargs": {
|
| 310 |
+
"max_gen_toks": 1280
|
| 311 |
+
},
|
| 312 |
+
"random_seed": 0,
|
| 313 |
+
"numpy_seed": 1234,
|
| 314 |
+
"torch_seed": 1234,
|
| 315 |
+
"fewshot_seed": 1234
|
| 316 |
+
},
|
| 317 |
+
"git_hash": "247c7f0",
|
| 318 |
+
"date": 1784324809.2486784,
|
| 319 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 320 |
+
"transformers_version": "4.57.6",
|
| 321 |
+
"lm_eval_version": "0.4.12",
|
| 322 |
+
"upper_git_hash": null,
|
| 323 |
+
"task_hashes": {
|
| 324 |
+
"gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
|
| 325 |
+
"minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
|
| 326 |
+
"ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
|
| 327 |
+
},
|
| 328 |
+
"model_source": "local-chat-completions",
|
| 329 |
+
"model_name": "student",
|
| 330 |
+
"model_name_sanitized": "student",
|
| 331 |
+
"system_instruction": null,
|
| 332 |
+
"system_instruction_sha": null,
|
| 333 |
+
"fewshot_as_multiturn": true,
|
| 334 |
+
"chat_template": "",
|
| 335 |
+
"chat_template_sha": null,
|
| 336 |
+
"total_evaluation_time_seconds": "599.8717604710109"
|
| 337 |
+
}
|
evals/general_suite/healed/glean_keep25_s1224/student/results_2026-07-17T15-01-00.225834.json
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"humaneval": {
|
| 4 |
+
"name": "humaneval",
|
| 5 |
+
"alias": "humaneval",
|
| 6 |
+
"sample_len": 164,
|
| 7 |
+
"pass@1,create_test": 0.018292682926829267,
|
| 8 |
+
"pass@1_stderr,create_test": 0.010496292269168307
|
| 9 |
+
},
|
| 10 |
+
"mbpp": {
|
| 11 |
+
"name": "mbpp",
|
| 12 |
+
"alias": "mbpp",
|
| 13 |
+
"sample_len": 500,
|
| 14 |
+
"pass_at_1,none": 0.094,
|
| 15 |
+
"pass_at_1_stderr,none": 0.013064047561049175
|
| 16 |
+
}
|
| 17 |
+
},
|
| 18 |
+
"group_subtasks": {},
|
| 19 |
+
"configs": {
|
| 20 |
+
"humaneval": {
|
| 21 |
+
"task": "humaneval",
|
| 22 |
+
"dataset_path": "openai/openai_humaneval",
|
| 23 |
+
"test_split": "test",
|
| 24 |
+
"doc_to_text": "{{prompt}}",
|
| 25 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 26 |
+
"unsafe_code": true,
|
| 27 |
+
"description": "",
|
| 28 |
+
"target_delimiter": " ",
|
| 29 |
+
"fewshot_delimiter": "\n\n",
|
| 30 |
+
"fewshot_config": {
|
| 31 |
+
"sampler": "default",
|
| 32 |
+
"split": null,
|
| 33 |
+
"process_docs": null,
|
| 34 |
+
"fewshot_indices": null,
|
| 35 |
+
"samples": null,
|
| 36 |
+
"doc_to_text": "{{prompt}}",
|
| 37 |
+
"doc_to_choice": null,
|
| 38 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 39 |
+
"gen_prefix": null,
|
| 40 |
+
"fewshot_delimiter": "\n\n",
|
| 41 |
+
"target_delimiter": " "
|
| 42 |
+
},
|
| 43 |
+
"num_fewshot": 0,
|
| 44 |
+
"metric_list": [
|
| 45 |
+
{
|
| 46 |
+
"metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
|
| 47 |
+
"aggregation": "mean",
|
| 48 |
+
"higher_is_better": true,
|
| 49 |
+
"k": [
|
| 50 |
+
1
|
| 51 |
+
]
|
| 52 |
+
}
|
| 53 |
+
],
|
| 54 |
+
"output_type": "generate_until",
|
| 55 |
+
"generation_kwargs": {
|
| 56 |
+
"until": [
|
| 57 |
+
"\nclass",
|
| 58 |
+
"\ndef",
|
| 59 |
+
"\n#",
|
| 60 |
+
"\nif",
|
| 61 |
+
"\nprint"
|
| 62 |
+
],
|
| 63 |
+
"max_gen_toks": 1024,
|
| 64 |
+
"do_sample": false
|
| 65 |
+
},
|
| 66 |
+
"repeats": 1,
|
| 67 |
+
"filter_list": [
|
| 68 |
+
{
|
| 69 |
+
"name": "create_test",
|
| 70 |
+
"filter": [
|
| 71 |
+
{
|
| 72 |
+
"function": "custom",
|
| 73 |
+
"filter_fn": "<function build_predictions at 0x7474b627eca0>"
|
| 74 |
+
}
|
| 75 |
+
]
|
| 76 |
+
}
|
| 77 |
+
],
|
| 78 |
+
"should_decontaminate": false,
|
| 79 |
+
"metadata": {
|
| 80 |
+
"version": 1.0,
|
| 81 |
+
"model": "student",
|
| 82 |
+
"base_url": "http://127.0.0.1:8420/v1/completions",
|
| 83 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep25_s1224/step0150",
|
| 84 |
+
"num_concurrent": 48,
|
| 85 |
+
"tokenized_requests": false,
|
| 86 |
+
"max_retries": 3,
|
| 87 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
|
| 88 |
+
}
|
| 89 |
+
},
|
| 90 |
+
"mbpp": {
|
| 91 |
+
"task": "mbpp",
|
| 92 |
+
"dataset_path": "google-research-datasets/mbpp",
|
| 93 |
+
"dataset_name": "full",
|
| 94 |
+
"test_split": "test",
|
| 95 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 96 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 97 |
+
"unsafe_code": true,
|
| 98 |
+
"description": "",
|
| 99 |
+
"target_delimiter": "",
|
| 100 |
+
"fewshot_delimiter": "\n\n",
|
| 101 |
+
"fewshot_config": {
|
| 102 |
+
"sampler": "first_n",
|
| 103 |
+
"split": null,
|
| 104 |
+
"process_docs": null,
|
| 105 |
+
"fewshot_indices": null,
|
| 106 |
+
"samples": "<function list_fewshot_samples at 0x74757f7679c0>",
|
| 107 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 108 |
+
"doc_to_choice": null,
|
| 109 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 110 |
+
"gen_prefix": null,
|
| 111 |
+
"fewshot_delimiter": "\n\n",
|
| 112 |
+
"target_delimiter": ""
|
| 113 |
+
},
|
| 114 |
+
"num_fewshot": 3,
|
| 115 |
+
"metric_list": [
|
| 116 |
+
{
|
| 117 |
+
"metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
|
| 118 |
+
"aggregation": "mean",
|
| 119 |
+
"higher_is_better": true
|
| 120 |
+
}
|
| 121 |
+
],
|
| 122 |
+
"output_type": "generate_until",
|
| 123 |
+
"generation_kwargs": {
|
| 124 |
+
"until": [
|
| 125 |
+
"[DONE]"
|
| 126 |
+
],
|
| 127 |
+
"do_sample": false
|
| 128 |
+
},
|
| 129 |
+
"repeats": 1,
|
| 130 |
+
"should_decontaminate": false,
|
| 131 |
+
"metadata": {
|
| 132 |
+
"version": 1.0,
|
| 133 |
+
"model": "student",
|
| 134 |
+
"base_url": "http://127.0.0.1:8420/v1/completions",
|
| 135 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep25_s1224/step0150",
|
| 136 |
+
"num_concurrent": 48,
|
| 137 |
+
"tokenized_requests": false,
|
| 138 |
+
"max_retries": 3,
|
| 139 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
|
| 140 |
+
}
|
| 141 |
+
}
|
| 142 |
+
},
|
| 143 |
+
"versions": {
|
| 144 |
+
"humaneval": 1.0,
|
| 145 |
+
"mbpp": 1.0
|
| 146 |
+
},
|
| 147 |
+
"n-shot": {
|
| 148 |
+
"humaneval": 0,
|
| 149 |
+
"mbpp": 3
|
| 150 |
+
},
|
| 151 |
+
"higher_is_better": {
|
| 152 |
+
"humaneval": {
|
| 153 |
+
"pass_at_k": true,
|
| 154 |
+
"pass@1": true
|
| 155 |
+
},
|
| 156 |
+
"mbpp": {
|
| 157 |
+
"pass_at_1": true
|
| 158 |
+
}
|
| 159 |
+
},
|
| 160 |
+
"n-samples": {
|
| 161 |
+
"humaneval": {
|
| 162 |
+
"original": 164,
|
| 163 |
+
"effective": 164
|
| 164 |
+
},
|
| 165 |
+
"mbpp": {
|
| 166 |
+
"original": 500,
|
| 167 |
+
"effective": 500
|
| 168 |
+
}
|
| 169 |
+
},
|
| 170 |
+
"config": {
|
| 171 |
+
"model": "local-completions",
|
| 172 |
+
"model_args": {
|
| 173 |
+
"model": "student",
|
| 174 |
+
"base_url": "http://127.0.0.1:8420/v1/completions",
|
| 175 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep25_s1224/step0150",
|
| 176 |
+
"num_concurrent": 48,
|
| 177 |
+
"tokenized_requests": false,
|
| 178 |
+
"max_retries": 3
|
| 179 |
+
},
|
| 180 |
+
"batch_size": 1,
|
| 181 |
+
"batch_sizes": [],
|
| 182 |
+
"device": "cuda:0",
|
| 183 |
+
"use_cache": null,
|
| 184 |
+
"limit": null,
|
| 185 |
+
"bootstrap_iters": 100000,
|
| 186 |
+
"gen_kwargs": {},
|
| 187 |
+
"random_seed": 0,
|
| 188 |
+
"numpy_seed": 1234,
|
| 189 |
+
"torch_seed": 1234,
|
| 190 |
+
"fewshot_seed": 1234
|
| 191 |
+
},
|
| 192 |
+
"git_hash": "247c7f0",
|
| 193 |
+
"date": 1784325412.1796849,
|
| 194 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 195 |
+
"transformers_version": "4.57.6",
|
| 196 |
+
"lm_eval_version": "0.4.12",
|
| 197 |
+
"upper_git_hash": null,
|
| 198 |
+
"tokenizer_pad_token": [
|
| 199 |
+
"<pad>",
|
| 200 |
+
"50280"
|
| 201 |
+
],
|
| 202 |
+
"tokenizer_eos_token": [
|
| 203 |
+
"|||IP_ADDRESS|||",
|
| 204 |
+
"50279"
|
| 205 |
+
],
|
| 206 |
+
"tokenizer_bos_token": [
|
| 207 |
+
"|||IP_ADDRESS|||",
|
| 208 |
+
"50279"
|
| 209 |
+
],
|
| 210 |
+
"eot_token_id": 50279,
|
| 211 |
+
"max_length": 2047,
|
| 212 |
+
"task_hashes": {
|
| 213 |
+
"humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
|
| 214 |
+
"mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
|
| 215 |
+
},
|
| 216 |
+
"model_source": "local-completions",
|
| 217 |
+
"model_name": "student",
|
| 218 |
+
"model_name_sanitized": "student",
|
| 219 |
+
"system_instruction": null,
|
| 220 |
+
"system_instruction_sha": null,
|
| 221 |
+
"fewshot_as_multiturn": null,
|
| 222 |
+
"chat_template": null,
|
| 223 |
+
"chat_template_sha": null,
|
| 224 |
+
"total_evaluation_time_seconds": "254.3856012949982"
|
| 225 |
+
}
|
evals/general_suite/healed/glean_keep25_s1224/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-56-42.779609.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1224/student/samples_humaneval_2026-07-17T15-01-00.225834.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1224/student/samples_ifeval_2026-07-17T14-56-42.779609.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1224/student/samples_mbpp_2026-07-17T15-01-00.225834.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1224/student/samples_minerva_math500_2026-07-17T14-56-42.779609.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1225/student/results_2026-07-17T14-48-07.537542.json
ADDED
|
@@ -0,0 +1,337 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"gsm8k_cot_zeroshot": {
|
| 4 |
+
"name": "gsm8k_cot_zeroshot",
|
| 5 |
+
"alias": "gsm8k_cot_zeroshot",
|
| 6 |
+
"sample_len": 1319,
|
| 7 |
+
"exact_match,strict-match": 0.0,
|
| 8 |
+
"exact_match_stderr,strict-match": 0.0,
|
| 9 |
+
"exact_match,flexible-extract": 0.21531463229719486,
|
| 10 |
+
"exact_match_stderr,flexible-extract": 0.01132209629457967
|
| 11 |
+
},
|
| 12 |
+
"minerva_math500": {
|
| 13 |
+
"name": "minerva_math500",
|
| 14 |
+
"alias": "minerva_math500",
|
| 15 |
+
"sample_len": 500,
|
| 16 |
+
"exact_match,none": 0.034,
|
| 17 |
+
"exact_match_stderr,none": 0.008112930375552172,
|
| 18 |
+
"math_verify,none": 0.088,
|
| 19 |
+
"math_verify_stderr,none": 0.012682021421471788
|
| 20 |
+
},
|
| 21 |
+
"ifeval": {
|
| 22 |
+
"name": "ifeval",
|
| 23 |
+
"alias": "ifeval",
|
| 24 |
+
"sample_len": 541,
|
| 25 |
+
"prompt_level_strict_acc,none": 0.4232902033271719,
|
| 26 |
+
"prompt_level_strict_acc_stderr,none": 0.021261842325248494,
|
| 27 |
+
"inst_level_strict_acc,none": 0.5563549160671463,
|
| 28 |
+
"inst_level_strict_acc_stderr,none": "N/A",
|
| 29 |
+
"prompt_level_loose_acc,none": 0.4565619223659889,
|
| 30 |
+
"prompt_level_loose_acc_stderr,none": 0.021435222545538937,
|
| 31 |
+
"inst_level_loose_acc,none": 0.5875299760191847,
|
| 32 |
+
"inst_level_loose_acc_stderr,none": "N/A"
|
| 33 |
+
}
|
| 34 |
+
},
|
| 35 |
+
"group_subtasks": {},
|
| 36 |
+
"configs": {
|
| 37 |
+
"gsm8k_cot_zeroshot": {
|
| 38 |
+
"task": "gsm8k_cot_zeroshot",
|
| 39 |
+
"dataset_path": "openai/gsm8k",
|
| 40 |
+
"dataset_name": "main",
|
| 41 |
+
"training_split": "train",
|
| 42 |
+
"test_split": "test",
|
| 43 |
+
"fewshot_split": "train",
|
| 44 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 45 |
+
"doc_to_target": "{{answer}}",
|
| 46 |
+
"unsafe_code": false,
|
| 47 |
+
"description": "",
|
| 48 |
+
"target_delimiter": " ",
|
| 49 |
+
"fewshot_delimiter": "\n\n",
|
| 50 |
+
"fewshot_config": {
|
| 51 |
+
"sampler": "default",
|
| 52 |
+
"split": "train",
|
| 53 |
+
"process_docs": null,
|
| 54 |
+
"fewshot_indices": null,
|
| 55 |
+
"samples": null,
|
| 56 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 57 |
+
"doc_to_choice": null,
|
| 58 |
+
"doc_to_target": "{{answer}}",
|
| 59 |
+
"gen_prefix": null,
|
| 60 |
+
"fewshot_delimiter": "\n\n",
|
| 61 |
+
"target_delimiter": " "
|
| 62 |
+
},
|
| 63 |
+
"num_fewshot": 0,
|
| 64 |
+
"metric_list": [
|
| 65 |
+
{
|
| 66 |
+
"metric": "exact_match",
|
| 67 |
+
"aggregation": "mean",
|
| 68 |
+
"higher_is_better": true,
|
| 69 |
+
"ignore_case": true,
|
| 70 |
+
"ignore_punctuation": false,
|
| 71 |
+
"regexes_to_ignore": [
|
| 72 |
+
",",
|
| 73 |
+
"\\$",
|
| 74 |
+
"(?s).*#### ",
|
| 75 |
+
"\\.$"
|
| 76 |
+
]
|
| 77 |
+
}
|
| 78 |
+
],
|
| 79 |
+
"output_type": "generate_until",
|
| 80 |
+
"generation_kwargs": {
|
| 81 |
+
"until": [
|
| 82 |
+
"Q:",
|
| 83 |
+
"</s>",
|
| 84 |
+
"<|im_end|>"
|
| 85 |
+
],
|
| 86 |
+
"do_sample": false,
|
| 87 |
+
"max_gen_toks": 1280
|
| 88 |
+
},
|
| 89 |
+
"repeats": 1,
|
| 90 |
+
"filter_list": [
|
| 91 |
+
{
|
| 92 |
+
"name": "strict-match",
|
| 93 |
+
"filter": [
|
| 94 |
+
{
|
| 95 |
+
"function": "regex",
|
| 96 |
+
"regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"function": "take_first"
|
| 100 |
+
}
|
| 101 |
+
]
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"name": "flexible-extract",
|
| 105 |
+
"filter": [
|
| 106 |
+
{
|
| 107 |
+
"function": "regex",
|
| 108 |
+
"group_select": -1,
|
| 109 |
+
"regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"function": "take_first"
|
| 113 |
+
}
|
| 114 |
+
]
|
| 115 |
+
}
|
| 116 |
+
],
|
| 117 |
+
"should_decontaminate": false,
|
| 118 |
+
"metadata": {
|
| 119 |
+
"version": 3.0,
|
| 120 |
+
"model": "student",
|
| 121 |
+
"base_url": "http://127.0.0.1:8421/v1/chat/completions",
|
| 122 |
+
"num_concurrent": 48,
|
| 123 |
+
"tokenized_requests": false,
|
| 124 |
+
"max_retries": 3,
|
| 125 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
|
| 126 |
+
}
|
| 127 |
+
},
|
| 128 |
+
"ifeval": {
|
| 129 |
+
"task": "ifeval",
|
| 130 |
+
"dataset_path": "google/IFEval",
|
| 131 |
+
"test_split": "train",
|
| 132 |
+
"doc_to_text": "prompt",
|
| 133 |
+
"doc_to_target": 0,
|
| 134 |
+
"unsafe_code": false,
|
| 135 |
+
"process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
|
| 136 |
+
"description": "",
|
| 137 |
+
"target_delimiter": " ",
|
| 138 |
+
"fewshot_delimiter": "\n\n",
|
| 139 |
+
"fewshot_config": {
|
| 140 |
+
"sampler": "default",
|
| 141 |
+
"split": null,
|
| 142 |
+
"process_docs": null,
|
| 143 |
+
"fewshot_indices": null,
|
| 144 |
+
"samples": null,
|
| 145 |
+
"doc_to_text": "prompt",
|
| 146 |
+
"doc_to_choice": null,
|
| 147 |
+
"doc_to_target": 0,
|
| 148 |
+
"gen_prefix": null,
|
| 149 |
+
"fewshot_delimiter": "\n\n",
|
| 150 |
+
"target_delimiter": " "
|
| 151 |
+
},
|
| 152 |
+
"num_fewshot": 0,
|
| 153 |
+
"metric_list": [
|
| 154 |
+
{
|
| 155 |
+
"metric": "prompt_level_strict_acc",
|
| 156 |
+
"aggregation": "mean",
|
| 157 |
+
"higher_is_better": true
|
| 158 |
+
},
|
| 159 |
+
{
|
| 160 |
+
"metric": "inst_level_strict_acc",
|
| 161 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 162 |
+
"higher_is_better": true
|
| 163 |
+
},
|
| 164 |
+
{
|
| 165 |
+
"metric": "prompt_level_loose_acc",
|
| 166 |
+
"aggregation": "mean",
|
| 167 |
+
"higher_is_better": true
|
| 168 |
+
},
|
| 169 |
+
{
|
| 170 |
+
"metric": "inst_level_loose_acc",
|
| 171 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 172 |
+
"higher_is_better": true
|
| 173 |
+
}
|
| 174 |
+
],
|
| 175 |
+
"output_type": "generate_until",
|
| 176 |
+
"generation_kwargs": {
|
| 177 |
+
"until": [],
|
| 178 |
+
"do_sample": false,
|
| 179 |
+
"temperature": 0.0,
|
| 180 |
+
"max_gen_toks": 1280
|
| 181 |
+
},
|
| 182 |
+
"repeats": 1,
|
| 183 |
+
"should_decontaminate": false,
|
| 184 |
+
"metadata": {
|
| 185 |
+
"version": 4.0,
|
| 186 |
+
"model": "student",
|
| 187 |
+
"base_url": "http://127.0.0.1:8421/v1/chat/completions",
|
| 188 |
+
"num_concurrent": 48,
|
| 189 |
+
"tokenized_requests": false,
|
| 190 |
+
"max_retries": 3,
|
| 191 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
|
| 192 |
+
}
|
| 193 |
+
},
|
| 194 |
+
"minerva_math500": {
|
| 195 |
+
"task": "minerva_math500",
|
| 196 |
+
"dataset_path": "HuggingFaceH4/MATH-500",
|
| 197 |
+
"dataset_name": "default",
|
| 198 |
+
"test_split": "test",
|
| 199 |
+
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
|
| 200 |
+
"doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
|
| 201 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 202 |
+
"unsafe_code": false,
|
| 203 |
+
"process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
|
| 204 |
+
"description": "",
|
| 205 |
+
"target_delimiter": " ",
|
| 206 |
+
"fewshot_delimiter": "\n\n",
|
| 207 |
+
"fewshot_config": {
|
| 208 |
+
"sampler": "first_n",
|
| 209 |
+
"split": null,
|
| 210 |
+
"process_docs": "<function process_docs at 0x78d5b8035b20>",
|
| 211 |
+
"fewshot_indices": null,
|
| 212 |
+
"samples": "<function list_fewshot_samples at 0x78d5b80379c0>",
|
| 213 |
+
"doc_to_text": "<function doc_to_text at 0x78d5b97fd8a0>",
|
| 214 |
+
"doc_to_choice": null,
|
| 215 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 216 |
+
"gen_prefix": null,
|
| 217 |
+
"fewshot_delimiter": "\n\n",
|
| 218 |
+
"target_delimiter": " "
|
| 219 |
+
},
|
| 220 |
+
"num_fewshot": 4,
|
| 221 |
+
"metric_list": [
|
| 222 |
+
{
|
| 223 |
+
"metric": "exact_match",
|
| 224 |
+
"aggregation": "mean",
|
| 225 |
+
"higher_is_better": true
|
| 226 |
+
},
|
| 227 |
+
{
|
| 228 |
+
"metric": "math_verify",
|
| 229 |
+
"aggregation": "mean",
|
| 230 |
+
"higher_is_better": true
|
| 231 |
+
}
|
| 232 |
+
],
|
| 233 |
+
"output_type": "generate_until",
|
| 234 |
+
"generation_kwargs": {
|
| 235 |
+
"until": [
|
| 236 |
+
"Problem:"
|
| 237 |
+
],
|
| 238 |
+
"do_sample": false,
|
| 239 |
+
"temperature": 0.0,
|
| 240 |
+
"max_gen_toks": 1280
|
| 241 |
+
},
|
| 242 |
+
"repeats": 1,
|
| 243 |
+
"should_decontaminate": false,
|
| 244 |
+
"metadata": {
|
| 245 |
+
"version": 3.0,
|
| 246 |
+
"model": "student",
|
| 247 |
+
"base_url": "http://127.0.0.1:8421/v1/chat/completions",
|
| 248 |
+
"num_concurrent": 48,
|
| 249 |
+
"tokenized_requests": false,
|
| 250 |
+
"max_retries": 3,
|
| 251 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
|
| 252 |
+
}
|
| 253 |
+
}
|
| 254 |
+
},
|
| 255 |
+
"versions": {
|
| 256 |
+
"gsm8k_cot_zeroshot": 3.0,
|
| 257 |
+
"ifeval": 4.0,
|
| 258 |
+
"minerva_math500": 3.0
|
| 259 |
+
},
|
| 260 |
+
"n-shot": {
|
| 261 |
+
"gsm8k_cot_zeroshot": 0,
|
| 262 |
+
"ifeval": 0,
|
| 263 |
+
"minerva_math500": 4
|
| 264 |
+
},
|
| 265 |
+
"higher_is_better": {
|
| 266 |
+
"gsm8k_cot_zeroshot": {
|
| 267 |
+
"exact_match": true
|
| 268 |
+
},
|
| 269 |
+
"ifeval": {
|
| 270 |
+
"prompt_level_strict_acc": true,
|
| 271 |
+
"inst_level_strict_acc": true,
|
| 272 |
+
"prompt_level_loose_acc": true,
|
| 273 |
+
"inst_level_loose_acc": true
|
| 274 |
+
},
|
| 275 |
+
"minerva_math500": {
|
| 276 |
+
"exact_match": true,
|
| 277 |
+
"math_verify": true
|
| 278 |
+
}
|
| 279 |
+
},
|
| 280 |
+
"n-samples": {
|
| 281 |
+
"gsm8k_cot_zeroshot": {
|
| 282 |
+
"original": 1319,
|
| 283 |
+
"effective": 1319
|
| 284 |
+
},
|
| 285 |
+
"minerva_math500": {
|
| 286 |
+
"original": 500,
|
| 287 |
+
"effective": 500
|
| 288 |
+
},
|
| 289 |
+
"ifeval": {
|
| 290 |
+
"original": 541,
|
| 291 |
+
"effective": 541
|
| 292 |
+
}
|
| 293 |
+
},
|
| 294 |
+
"config": {
|
| 295 |
+
"model": "local-chat-completions",
|
| 296 |
+
"model_args": {
|
| 297 |
+
"model": "student",
|
| 298 |
+
"base_url": "http://127.0.0.1:8421/v1/chat/completions",
|
| 299 |
+
"num_concurrent": 48,
|
| 300 |
+
"tokenized_requests": false,
|
| 301 |
+
"max_retries": 3
|
| 302 |
+
},
|
| 303 |
+
"batch_size": 1,
|
| 304 |
+
"batch_sizes": [],
|
| 305 |
+
"device": "cuda:0",
|
| 306 |
+
"use_cache": null,
|
| 307 |
+
"limit": null,
|
| 308 |
+
"bootstrap_iters": 100000,
|
| 309 |
+
"gen_kwargs": {
|
| 310 |
+
"max_gen_toks": 1280
|
| 311 |
+
},
|
| 312 |
+
"random_seed": 0,
|
| 313 |
+
"numpy_seed": 1234,
|
| 314 |
+
"torch_seed": 1234,
|
| 315 |
+
"fewshot_seed": 1234
|
| 316 |
+
},
|
| 317 |
+
"git_hash": "247c7f0",
|
| 318 |
+
"date": 1784324273.5688035,
|
| 319 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 320 |
+
"transformers_version": "4.57.6",
|
| 321 |
+
"lm_eval_version": "0.4.12",
|
| 322 |
+
"upper_git_hash": null,
|
| 323 |
+
"task_hashes": {
|
| 324 |
+
"gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
|
| 325 |
+
"minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
|
| 326 |
+
"ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
|
| 327 |
+
},
|
| 328 |
+
"model_source": "local-chat-completions",
|
| 329 |
+
"model_name": "student",
|
| 330 |
+
"model_name_sanitized": "student",
|
| 331 |
+
"system_instruction": null,
|
| 332 |
+
"system_instruction_sha": null,
|
| 333 |
+
"fewshot_as_multiturn": true,
|
| 334 |
+
"chat_template": "",
|
| 335 |
+
"chat_template_sha": null,
|
| 336 |
+
"total_evaluation_time_seconds": "620.2983292259887"
|
| 337 |
+
}
|
evals/general_suite/healed/glean_keep25_s1225/student/results_2026-07-17T14-54-29.755565.json
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"humaneval": {
|
| 4 |
+
"name": "humaneval",
|
| 5 |
+
"alias": "humaneval",
|
| 6 |
+
"sample_len": 164,
|
| 7 |
+
"pass@1,create_test": 0.018292682926829267,
|
| 8 |
+
"pass@1_stderr,create_test": 0.010496292269168307
|
| 9 |
+
},
|
| 10 |
+
"mbpp": {
|
| 11 |
+
"name": "mbpp",
|
| 12 |
+
"alias": "mbpp",
|
| 13 |
+
"sample_len": 500,
|
| 14 |
+
"pass_at_1,none": 0.132,
|
| 15 |
+
"pass_at_1_stderr,none": 0.01515292785058016
|
| 16 |
+
}
|
| 17 |
+
},
|
| 18 |
+
"group_subtasks": {},
|
| 19 |
+
"configs": {
|
| 20 |
+
"humaneval": {
|
| 21 |
+
"task": "humaneval",
|
| 22 |
+
"dataset_path": "openai/openai_humaneval",
|
| 23 |
+
"test_split": "test",
|
| 24 |
+
"doc_to_text": "{{prompt}}",
|
| 25 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 26 |
+
"unsafe_code": true,
|
| 27 |
+
"description": "",
|
| 28 |
+
"target_delimiter": " ",
|
| 29 |
+
"fewshot_delimiter": "\n\n",
|
| 30 |
+
"fewshot_config": {
|
| 31 |
+
"sampler": "default",
|
| 32 |
+
"split": null,
|
| 33 |
+
"process_docs": null,
|
| 34 |
+
"fewshot_indices": null,
|
| 35 |
+
"samples": null,
|
| 36 |
+
"doc_to_text": "{{prompt}}",
|
| 37 |
+
"doc_to_choice": null,
|
| 38 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 39 |
+
"gen_prefix": null,
|
| 40 |
+
"fewshot_delimiter": "\n\n",
|
| 41 |
+
"target_delimiter": " "
|
| 42 |
+
},
|
| 43 |
+
"num_fewshot": 0,
|
| 44 |
+
"metric_list": [
|
| 45 |
+
{
|
| 46 |
+
"metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
|
| 47 |
+
"aggregation": "mean",
|
| 48 |
+
"higher_is_better": true,
|
| 49 |
+
"k": [
|
| 50 |
+
1
|
| 51 |
+
]
|
| 52 |
+
}
|
| 53 |
+
],
|
| 54 |
+
"output_type": "generate_until",
|
| 55 |
+
"generation_kwargs": {
|
| 56 |
+
"until": [
|
| 57 |
+
"\nclass",
|
| 58 |
+
"\ndef",
|
| 59 |
+
"\n#",
|
| 60 |
+
"\nif",
|
| 61 |
+
"\nprint"
|
| 62 |
+
],
|
| 63 |
+
"max_gen_toks": 1024,
|
| 64 |
+
"do_sample": false
|
| 65 |
+
},
|
| 66 |
+
"repeats": 1,
|
| 67 |
+
"filter_list": [
|
| 68 |
+
{
|
| 69 |
+
"name": "create_test",
|
| 70 |
+
"filter": [
|
| 71 |
+
{
|
| 72 |
+
"function": "custom",
|
| 73 |
+
"filter_fn": "<function build_predictions at 0x7ceb8c426ca0>"
|
| 74 |
+
}
|
| 75 |
+
]
|
| 76 |
+
}
|
| 77 |
+
],
|
| 78 |
+
"should_decontaminate": false,
|
| 79 |
+
"metadata": {
|
| 80 |
+
"version": 1.0,
|
| 81 |
+
"model": "student",
|
| 82 |
+
"base_url": "http://127.0.0.1:8421/v1/completions",
|
| 83 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep25_s1225/step0150",
|
| 84 |
+
"num_concurrent": 48,
|
| 85 |
+
"tokenized_requests": false,
|
| 86 |
+
"max_retries": 3,
|
| 87 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
|
| 88 |
+
}
|
| 89 |
+
},
|
| 90 |
+
"mbpp": {
|
| 91 |
+
"task": "mbpp",
|
| 92 |
+
"dataset_path": "google-research-datasets/mbpp",
|
| 93 |
+
"dataset_name": "full",
|
| 94 |
+
"test_split": "test",
|
| 95 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 96 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 97 |
+
"unsafe_code": true,
|
| 98 |
+
"description": "",
|
| 99 |
+
"target_delimiter": "",
|
| 100 |
+
"fewshot_delimiter": "\n\n",
|
| 101 |
+
"fewshot_config": {
|
| 102 |
+
"sampler": "first_n",
|
| 103 |
+
"split": null,
|
| 104 |
+
"process_docs": null,
|
| 105 |
+
"fewshot_indices": null,
|
| 106 |
+
"samples": "<function list_fewshot_samples at 0x7cec555079c0>",
|
| 107 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 108 |
+
"doc_to_choice": null,
|
| 109 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 110 |
+
"gen_prefix": null,
|
| 111 |
+
"fewshot_delimiter": "\n\n",
|
| 112 |
+
"target_delimiter": ""
|
| 113 |
+
},
|
| 114 |
+
"num_fewshot": 3,
|
| 115 |
+
"metric_list": [
|
| 116 |
+
{
|
| 117 |
+
"metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
|
| 118 |
+
"aggregation": "mean",
|
| 119 |
+
"higher_is_better": true
|
| 120 |
+
}
|
| 121 |
+
],
|
| 122 |
+
"output_type": "generate_until",
|
| 123 |
+
"generation_kwargs": {
|
| 124 |
+
"until": [
|
| 125 |
+
"[DONE]"
|
| 126 |
+
],
|
| 127 |
+
"do_sample": false
|
| 128 |
+
},
|
| 129 |
+
"repeats": 1,
|
| 130 |
+
"should_decontaminate": false,
|
| 131 |
+
"metadata": {
|
| 132 |
+
"version": 1.0,
|
| 133 |
+
"model": "student",
|
| 134 |
+
"base_url": "http://127.0.0.1:8421/v1/completions",
|
| 135 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep25_s1225/step0150",
|
| 136 |
+
"num_concurrent": 48,
|
| 137 |
+
"tokenized_requests": false,
|
| 138 |
+
"max_retries": 3,
|
| 139 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
|
| 140 |
+
}
|
| 141 |
+
}
|
| 142 |
+
},
|
| 143 |
+
"versions": {
|
| 144 |
+
"humaneval": 1.0,
|
| 145 |
+
"mbpp": 1.0
|
| 146 |
+
},
|
| 147 |
+
"n-shot": {
|
| 148 |
+
"humaneval": 0,
|
| 149 |
+
"mbpp": 3
|
| 150 |
+
},
|
| 151 |
+
"higher_is_better": {
|
| 152 |
+
"humaneval": {
|
| 153 |
+
"pass_at_k": true,
|
| 154 |
+
"pass@1": true
|
| 155 |
+
},
|
| 156 |
+
"mbpp": {
|
| 157 |
+
"pass_at_1": true
|
| 158 |
+
}
|
| 159 |
+
},
|
| 160 |
+
"n-samples": {
|
| 161 |
+
"humaneval": {
|
| 162 |
+
"original": 164,
|
| 163 |
+
"effective": 164
|
| 164 |
+
},
|
| 165 |
+
"mbpp": {
|
| 166 |
+
"original": 500,
|
| 167 |
+
"effective": 500
|
| 168 |
+
}
|
| 169 |
+
},
|
| 170 |
+
"config": {
|
| 171 |
+
"model": "local-completions",
|
| 172 |
+
"model_args": {
|
| 173 |
+
"model": "student",
|
| 174 |
+
"base_url": "http://127.0.0.1:8421/v1/completions",
|
| 175 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep25_s1225/step0150",
|
| 176 |
+
"num_concurrent": 48,
|
| 177 |
+
"tokenized_requests": false,
|
| 178 |
+
"max_retries": 3
|
| 179 |
+
},
|
| 180 |
+
"batch_size": 1,
|
| 181 |
+
"batch_sizes": [],
|
| 182 |
+
"device": "cuda:0",
|
| 183 |
+
"use_cache": null,
|
| 184 |
+
"limit": null,
|
| 185 |
+
"bootstrap_iters": 100000,
|
| 186 |
+
"gen_kwargs": {},
|
| 187 |
+
"random_seed": 0,
|
| 188 |
+
"numpy_seed": 1234,
|
| 189 |
+
"torch_seed": 1234,
|
| 190 |
+
"fewshot_seed": 1234
|
| 191 |
+
},
|
| 192 |
+
"git_hash": "247c7f0",
|
| 193 |
+
"date": 1784324897.002143,
|
| 194 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 195 |
+
"transformers_version": "4.57.6",
|
| 196 |
+
"lm_eval_version": "0.4.12",
|
| 197 |
+
"upper_git_hash": null,
|
| 198 |
+
"tokenizer_pad_token": [
|
| 199 |
+
"<pad>",
|
| 200 |
+
"50280"
|
| 201 |
+
],
|
| 202 |
+
"tokenizer_eos_token": [
|
| 203 |
+
"|||IP_ADDRESS|||",
|
| 204 |
+
"50279"
|
| 205 |
+
],
|
| 206 |
+
"tokenizer_bos_token": [
|
| 207 |
+
"|||IP_ADDRESS|||",
|
| 208 |
+
"50279"
|
| 209 |
+
],
|
| 210 |
+
"eot_token_id": 50279,
|
| 211 |
+
"max_length": 2047,
|
| 212 |
+
"task_hashes": {
|
| 213 |
+
"humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
|
| 214 |
+
"mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
|
| 215 |
+
},
|
| 216 |
+
"model_source": "local-completions",
|
| 217 |
+
"model_name": "student",
|
| 218 |
+
"model_name_sanitized": "student",
|
| 219 |
+
"system_instruction": null,
|
| 220 |
+
"system_instruction_sha": null,
|
| 221 |
+
"fewshot_as_multiturn": null,
|
| 222 |
+
"chat_template": null,
|
| 223 |
+
"chat_template_sha": null,
|
| 224 |
+
"total_evaluation_time_seconds": "379.11067679699045"
|
| 225 |
+
}
|
evals/general_suite/healed/glean_keep25_s1225/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-48-07.537542.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1225/student/samples_humaneval_2026-07-17T14-54-29.755565.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1225/student/samples_ifeval_2026-07-17T14-48-07.537542.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1225/student/samples_mbpp_2026-07-17T14-54-29.755565.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1225/student/samples_minerva_math500_2026-07-17T14-48-07.537542.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1226/lm_eval.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1226/lm_eval_code.log
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/164 [00:00<?, ?it/s]
|
| 1 |
89%|████████▉ | 146/164 [00:00<00:00, 1454.39it/s]
|
|
|
|
|
|
|
| 2 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 3 |
3%|▎ | 17/500 [00:00<00:02, 162.30it/s]
|
| 4 |
7%|▋ | 34/500 [00:00<00:02, 163.07it/s]
|
| 5 |
10%|█ | 51/500 [00:00<00:02, 163.33it/s]
|
| 6 |
14%|█▎ | 68/500 [00:00<00:02, 163.58it/s]
|
| 7 |
17%|█▋ | 85/500 [00:00<00:02, 163.87it/s]
|
| 8 |
20%|██ | 102/500 [00:00<00:02, 164.25it/s]
|
| 9 |
24%|██▍ | 119/500 [00:00<00:02, 164.37it/s]
|
| 10 |
27%|██▋ | 136/500 [00:00<00:02, 164.56it/s]
|
| 11 |
31%|███ | 153/500 [00:00<00:02, 164.42it/s]
|
| 12 |
34%|███▍ | 170/500 [00:01<00:02, 164.57it/s]
|
| 13 |
37%|███▋ | 187/500 [00:01<00:01, 164.80it/s]
|
| 14 |
41%|████ | 204/500 [00:01<00:01, 164.91it/s]
|
| 15 |
44%|████▍ | 221/500 [00:01<00:01, 165.06it/s]
|
| 16 |
48%|████▊ | 238/500 [00:01<00:01, 165.25it/s]
|
| 17 |
51%|█████ | 255/500 [00:01<00:01, 165.34it/s]
|
| 18 |
54%|█████▍ | 272/500 [00:01<00:01, 165.48it/s]
|
| 19 |
58%|█████▊ | 289/500 [00:01<00:01, 165.48it/s]
|
| 20 |
61%|██████ | 306/500 [00:01<00:01, 165.46it/s]
|
| 21 |
65%|██████▍ | 323/500 [00:01<00:01, 165.51it/s]
|
| 22 |
68%|██████▊ | 340/500 [00:02<00:00, 165.47it/s]
|
| 23 |
71%|███████▏ | 357/500 [00:02<00:00, 165.56it/s]
|
| 24 |
75%|███████▍ | 374/500 [00:02<00:00, 165.82it/s]
|
| 25 |
78%|███████▊ | 391/500 [00:02<00:00, 165.80it/s]
|
| 26 |
82%|████████▏ | 408/500 [00:02<00:00, 165.87it/s]
|
| 27 |
85%|████████▌ | 425/500 [00:02<00:00, 165.89it/s]
|
| 28 |
88%|████████▊ | 442/500 [00:02<00:00, 165.95it/s]
|
| 29 |
92%|█████████▏| 459/500 [00:02<00:00, 165.85it/s]
|
| 30 |
95%|█████████▌| 476/500 [00:02<00:00, 165.54it/s]
|
| 31 |
99%|█████████▊| 493/500 [00:02<00:00, 165.60it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17:14:48:43 INFO [_cli.run:388] Selected Tasks: ['humaneval', 'mbpp']
|
| 2 |
+
2026-07-17:14:48:44 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 3 |
+
2026-07-17:14:48:44 INFO [evaluator:239] Initializing local-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8422/v1/completions', 'tokenizer': 'outputs/healed/grid_general/glean_keep25_s1226/step0150', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 4 |
+
2026-07-17:14:48:44 INFO [models.openai_completions:42] Remote tokenizer not supported. Using huggingface tokenizer backend.
|
| 5 |
+
2026-07-17:14:48:44 INFO [models.api_models:179] Using max length 2048 - 1
|
| 6 |
+
2026-07-17:14:48:44 INFO [models.api_models:200] Using tokenizer huggingface
|
| 7 |
+
2026-07-17:14:48:51 INFO [evaluator_utils:446] Selected tasks:
|
| 8 |
+
2026-07-17:14:48:51 INFO [evaluator_utils:480] Task: humaneval (humaneval/humaneval.yaml)
|
| 9 |
+
2026-07-17:14:48:51 INFO [evaluator_utils:480] Task: mbpp (mbpp/mbpp.yaml)
|
| 10 |
+
2026-07-17:14:48:51 INFO [evaluator:314] humaneval: Using gen_kwargs: {'until': ['\nclass', '\ndef', '\n#', '\nif', '\nprint'], 'max_gen_toks': 1024, 'do_sample': False}
|
| 11 |
+
2026-07-17:14:48:51 INFO [evaluator:314] mbpp: Using gen_kwargs: {'until': ['[DONE]'], 'do_sample': False}
|
| 12 |
+
2026-07-17:14:48:51 INFO [api.task:312] Building contexts for humaneval on rank 0...
|
| 13 |
+
|
| 14 |
0%| | 0/164 [00:00<?, ?it/s]
|
| 15 |
89%|████████▉ | 146/164 [00:00<00:00, 1454.39it/s]
|
| 16 |
+
2026-07-17:14:48:51 INFO [api.task:312] Building contexts for mbpp on rank 0...
|
| 17 |
+
|
| 18 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 19 |
3%|▎ | 17/500 [00:00<00:02, 162.30it/s]
|
| 20 |
7%|▋ | 34/500 [00:00<00:02, 163.07it/s]
|
| 21 |
10%|█ | 51/500 [00:00<00:02, 163.33it/s]
|
| 22 |
14%|█▎ | 68/500 [00:00<00:02, 163.58it/s]
|
| 23 |
17%|█▋ | 85/500 [00:00<00:02, 163.87it/s]
|
| 24 |
20%|██ | 102/500 [00:00<00:02, 164.25it/s]
|
| 25 |
24%|██▍ | 119/500 [00:00<00:02, 164.37it/s]
|
| 26 |
27%|██▋ | 136/500 [00:00<00:02, 164.56it/s]
|
| 27 |
31%|███ | 153/500 [00:00<00:02, 164.42it/s]
|
| 28 |
34%|███▍ | 170/500 [00:01<00:02, 164.57it/s]
|
| 29 |
37%|███▋ | 187/500 [00:01<00:01, 164.80it/s]
|
| 30 |
41%|████ | 204/500 [00:01<00:01, 164.91it/s]
|
| 31 |
44%|████▍ | 221/500 [00:01<00:01, 165.06it/s]
|
| 32 |
48%|████▊ | 238/500 [00:01<00:01, 165.25it/s]
|
| 33 |
51%|█████ | 255/500 [00:01<00:01, 165.34it/s]
|
| 34 |
54%|█████▍ | 272/500 [00:01<00:01, 165.48it/s]
|
| 35 |
58%|█████▊ | 289/500 [00:01<00:01, 165.48it/s]
|
| 36 |
61%|██████ | 306/500 [00:01<00:01, 165.46it/s]
|
| 37 |
65%|██████▍ | 323/500 [00:01<00:01, 165.51it/s]
|
| 38 |
68%|██████▊ | 340/500 [00:02<00:00, 165.47it/s]
|
| 39 |
71%|███████▏ | 357/500 [00:02<00:00, 165.56it/s]
|
| 40 |
75%|███████▍ | 374/500 [00:02<00:00, 165.82it/s]
|
| 41 |
78%|███████▊ | 391/500 [00:02<00:00, 165.80it/s]
|
| 42 |
82%|████████▏ | 408/500 [00:02<00:00, 165.87it/s]
|
| 43 |
85%|████████▌ | 425/500 [00:02<00:00, 165.89it/s]
|
| 44 |
88%|████████▊ | 442/500 [00:02<00:00, 165.95it/s]
|
| 45 |
92%|█████████▏| 459/500 [00:02<00:00, 165.85it/s]
|
| 46 |
95%|█████████▌| 476/500 [00:02<00:00, 165.54it/s]
|
| 47 |
99%|█████████▊| 493/500 [00:02<00:00, 165.60it/s]
|
| 48 |
+
2026-07-17:14:48:54 INFO [evaluator:585] Running generate_until requests
|
| 49 |
+
2026-07-17:14:48:54 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
2026-07-17:14:54:44 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 53 |
+
2026-07-17:14:54:44 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/healed/glean_keep25_s1226/student/*.jsonl
|
| 54 |
+
local-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8422/v1/completions', 'tokenizer': 'outputs/healed/grid_general/glean_keep25_s1226/step0150', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: None, num_fewshot: None, batch_size: 1
|
| 55 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value | |Stderr|
|
| 56 |
+
|---------|------:|-----------|-----:|---------|---|-----:|---|-----:|
|
| 57 |
+
|humaneval| 1|create_test| 0|pass@1 |↑ |0.0244|± |0.0121|
|
| 58 |
+
|mbpp | 1|none | 3|pass_at_1|↑ |0.1100|± |0.0140|
|
| 59 |
+
|
evals/general_suite/healed/glean_keep25_s1226/server.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1226/student/results_2026-07-17T14-48-34.159099.json
ADDED
|
@@ -0,0 +1,337 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"gsm8k_cot_zeroshot": {
|
| 4 |
+
"name": "gsm8k_cot_zeroshot",
|
| 5 |
+
"alias": "gsm8k_cot_zeroshot",
|
| 6 |
+
"sample_len": 1319,
|
| 7 |
+
"exact_match,strict-match": 0.001516300227445034,
|
| 8 |
+
"exact_match_stderr,strict-match": 0.0010717793485492636,
|
| 9 |
+
"exact_match,flexible-extract": 0.21228203184230476,
|
| 10 |
+
"exact_match_stderr,flexible-extract": 0.011263783355400308
|
| 11 |
+
},
|
| 12 |
+
"minerva_math500": {
|
| 13 |
+
"name": "minerva_math500",
|
| 14 |
+
"alias": "minerva_math500",
|
| 15 |
+
"sample_len": 500,
|
| 16 |
+
"exact_match,none": 0.03,
|
| 17 |
+
"exact_match_stderr,none": 0.007636532803997706,
|
| 18 |
+
"math_verify,none": 0.084,
|
| 19 |
+
"math_verify_stderr,none": 0.012417584015643749
|
| 20 |
+
},
|
| 21 |
+
"ifeval": {
|
| 22 |
+
"name": "ifeval",
|
| 23 |
+
"alias": "ifeval",
|
| 24 |
+
"sample_len": 541,
|
| 25 |
+
"prompt_level_strict_acc,none": 0.4713493530499076,
|
| 26 |
+
"prompt_level_strict_acc_stderr,none": 0.02148122093008049,
|
| 27 |
+
"inst_level_strict_acc,none": 0.60431654676259,
|
| 28 |
+
"inst_level_strict_acc_stderr,none": "N/A",
|
| 29 |
+
"prompt_level_loose_acc,none": 0.5027726432532348,
|
| 30 |
+
"prompt_level_loose_acc_stderr,none": 0.021516243323548144,
|
| 31 |
+
"inst_level_loose_acc,none": 0.6282973621103117,
|
| 32 |
+
"inst_level_loose_acc_stderr,none": "N/A"
|
| 33 |
+
}
|
| 34 |
+
},
|
| 35 |
+
"group_subtasks": {},
|
| 36 |
+
"configs": {
|
| 37 |
+
"gsm8k_cot_zeroshot": {
|
| 38 |
+
"task": "gsm8k_cot_zeroshot",
|
| 39 |
+
"dataset_path": "openai/gsm8k",
|
| 40 |
+
"dataset_name": "main",
|
| 41 |
+
"training_split": "train",
|
| 42 |
+
"test_split": "test",
|
| 43 |
+
"fewshot_split": "train",
|
| 44 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 45 |
+
"doc_to_target": "{{answer}}",
|
| 46 |
+
"unsafe_code": false,
|
| 47 |
+
"description": "",
|
| 48 |
+
"target_delimiter": " ",
|
| 49 |
+
"fewshot_delimiter": "\n\n",
|
| 50 |
+
"fewshot_config": {
|
| 51 |
+
"sampler": "default",
|
| 52 |
+
"split": "train",
|
| 53 |
+
"process_docs": null,
|
| 54 |
+
"fewshot_indices": null,
|
| 55 |
+
"samples": null,
|
| 56 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 57 |
+
"doc_to_choice": null,
|
| 58 |
+
"doc_to_target": "{{answer}}",
|
| 59 |
+
"gen_prefix": null,
|
| 60 |
+
"fewshot_delimiter": "\n\n",
|
| 61 |
+
"target_delimiter": " "
|
| 62 |
+
},
|
| 63 |
+
"num_fewshot": 0,
|
| 64 |
+
"metric_list": [
|
| 65 |
+
{
|
| 66 |
+
"metric": "exact_match",
|
| 67 |
+
"aggregation": "mean",
|
| 68 |
+
"higher_is_better": true,
|
| 69 |
+
"ignore_case": true,
|
| 70 |
+
"ignore_punctuation": false,
|
| 71 |
+
"regexes_to_ignore": [
|
| 72 |
+
",",
|
| 73 |
+
"\\$",
|
| 74 |
+
"(?s).*#### ",
|
| 75 |
+
"\\.$"
|
| 76 |
+
]
|
| 77 |
+
}
|
| 78 |
+
],
|
| 79 |
+
"output_type": "generate_until",
|
| 80 |
+
"generation_kwargs": {
|
| 81 |
+
"until": [
|
| 82 |
+
"Q:",
|
| 83 |
+
"</s>",
|
| 84 |
+
"<|im_end|>"
|
| 85 |
+
],
|
| 86 |
+
"do_sample": false,
|
| 87 |
+
"max_gen_toks": 1280
|
| 88 |
+
},
|
| 89 |
+
"repeats": 1,
|
| 90 |
+
"filter_list": [
|
| 91 |
+
{
|
| 92 |
+
"name": "strict-match",
|
| 93 |
+
"filter": [
|
| 94 |
+
{
|
| 95 |
+
"function": "regex",
|
| 96 |
+
"regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"function": "take_first"
|
| 100 |
+
}
|
| 101 |
+
]
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"name": "flexible-extract",
|
| 105 |
+
"filter": [
|
| 106 |
+
{
|
| 107 |
+
"function": "regex",
|
| 108 |
+
"group_select": -1,
|
| 109 |
+
"regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"function": "take_first"
|
| 113 |
+
}
|
| 114 |
+
]
|
| 115 |
+
}
|
| 116 |
+
],
|
| 117 |
+
"should_decontaminate": false,
|
| 118 |
+
"metadata": {
|
| 119 |
+
"version": 3.0,
|
| 120 |
+
"model": "student",
|
| 121 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 122 |
+
"num_concurrent": 48,
|
| 123 |
+
"tokenized_requests": false,
|
| 124 |
+
"max_retries": 3,
|
| 125 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
|
| 126 |
+
}
|
| 127 |
+
},
|
| 128 |
+
"ifeval": {
|
| 129 |
+
"task": "ifeval",
|
| 130 |
+
"dataset_path": "google/IFEval",
|
| 131 |
+
"test_split": "train",
|
| 132 |
+
"doc_to_text": "prompt",
|
| 133 |
+
"doc_to_target": 0,
|
| 134 |
+
"unsafe_code": false,
|
| 135 |
+
"process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
|
| 136 |
+
"description": "",
|
| 137 |
+
"target_delimiter": " ",
|
| 138 |
+
"fewshot_delimiter": "\n\n",
|
| 139 |
+
"fewshot_config": {
|
| 140 |
+
"sampler": "default",
|
| 141 |
+
"split": null,
|
| 142 |
+
"process_docs": null,
|
| 143 |
+
"fewshot_indices": null,
|
| 144 |
+
"samples": null,
|
| 145 |
+
"doc_to_text": "prompt",
|
| 146 |
+
"doc_to_choice": null,
|
| 147 |
+
"doc_to_target": 0,
|
| 148 |
+
"gen_prefix": null,
|
| 149 |
+
"fewshot_delimiter": "\n\n",
|
| 150 |
+
"target_delimiter": " "
|
| 151 |
+
},
|
| 152 |
+
"num_fewshot": 0,
|
| 153 |
+
"metric_list": [
|
| 154 |
+
{
|
| 155 |
+
"metric": "prompt_level_strict_acc",
|
| 156 |
+
"aggregation": "mean",
|
| 157 |
+
"higher_is_better": true
|
| 158 |
+
},
|
| 159 |
+
{
|
| 160 |
+
"metric": "inst_level_strict_acc",
|
| 161 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 162 |
+
"higher_is_better": true
|
| 163 |
+
},
|
| 164 |
+
{
|
| 165 |
+
"metric": "prompt_level_loose_acc",
|
| 166 |
+
"aggregation": "mean",
|
| 167 |
+
"higher_is_better": true
|
| 168 |
+
},
|
| 169 |
+
{
|
| 170 |
+
"metric": "inst_level_loose_acc",
|
| 171 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 172 |
+
"higher_is_better": true
|
| 173 |
+
}
|
| 174 |
+
],
|
| 175 |
+
"output_type": "generate_until",
|
| 176 |
+
"generation_kwargs": {
|
| 177 |
+
"until": [],
|
| 178 |
+
"do_sample": false,
|
| 179 |
+
"temperature": 0.0,
|
| 180 |
+
"max_gen_toks": 1280
|
| 181 |
+
},
|
| 182 |
+
"repeats": 1,
|
| 183 |
+
"should_decontaminate": false,
|
| 184 |
+
"metadata": {
|
| 185 |
+
"version": 4.0,
|
| 186 |
+
"model": "student",
|
| 187 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 188 |
+
"num_concurrent": 48,
|
| 189 |
+
"tokenized_requests": false,
|
| 190 |
+
"max_retries": 3,
|
| 191 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
|
| 192 |
+
}
|
| 193 |
+
},
|
| 194 |
+
"minerva_math500": {
|
| 195 |
+
"task": "minerva_math500",
|
| 196 |
+
"dataset_path": "HuggingFaceH4/MATH-500",
|
| 197 |
+
"dataset_name": "default",
|
| 198 |
+
"test_split": "test",
|
| 199 |
+
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
|
| 200 |
+
"doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
|
| 201 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 202 |
+
"unsafe_code": false,
|
| 203 |
+
"process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
|
| 204 |
+
"description": "",
|
| 205 |
+
"target_delimiter": " ",
|
| 206 |
+
"fewshot_delimiter": "\n\n",
|
| 207 |
+
"fewshot_config": {
|
| 208 |
+
"sampler": "first_n",
|
| 209 |
+
"split": null,
|
| 210 |
+
"process_docs": "<function process_docs at 0x7b4197c31b20>",
|
| 211 |
+
"fewshot_indices": null,
|
| 212 |
+
"samples": "<function list_fewshot_samples at 0x7b4197c339c0>",
|
| 213 |
+
"doc_to_text": "<function doc_to_text at 0x7b419d3fd800>",
|
| 214 |
+
"doc_to_choice": null,
|
| 215 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 216 |
+
"gen_prefix": null,
|
| 217 |
+
"fewshot_delimiter": "\n\n",
|
| 218 |
+
"target_delimiter": " "
|
| 219 |
+
},
|
| 220 |
+
"num_fewshot": 4,
|
| 221 |
+
"metric_list": [
|
| 222 |
+
{
|
| 223 |
+
"metric": "exact_match",
|
| 224 |
+
"aggregation": "mean",
|
| 225 |
+
"higher_is_better": true
|
| 226 |
+
},
|
| 227 |
+
{
|
| 228 |
+
"metric": "math_verify",
|
| 229 |
+
"aggregation": "mean",
|
| 230 |
+
"higher_is_better": true
|
| 231 |
+
}
|
| 232 |
+
],
|
| 233 |
+
"output_type": "generate_until",
|
| 234 |
+
"generation_kwargs": {
|
| 235 |
+
"until": [
|
| 236 |
+
"Problem:"
|
| 237 |
+
],
|
| 238 |
+
"do_sample": false,
|
| 239 |
+
"temperature": 0.0,
|
| 240 |
+
"max_gen_toks": 1280
|
| 241 |
+
},
|
| 242 |
+
"repeats": 1,
|
| 243 |
+
"should_decontaminate": false,
|
| 244 |
+
"metadata": {
|
| 245 |
+
"version": 3.0,
|
| 246 |
+
"model": "student",
|
| 247 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 248 |
+
"num_concurrent": 48,
|
| 249 |
+
"tokenized_requests": false,
|
| 250 |
+
"max_retries": 3,
|
| 251 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
|
| 252 |
+
}
|
| 253 |
+
}
|
| 254 |
+
},
|
| 255 |
+
"versions": {
|
| 256 |
+
"gsm8k_cot_zeroshot": 3.0,
|
| 257 |
+
"ifeval": 4.0,
|
| 258 |
+
"minerva_math500": 3.0
|
| 259 |
+
},
|
| 260 |
+
"n-shot": {
|
| 261 |
+
"gsm8k_cot_zeroshot": 0,
|
| 262 |
+
"ifeval": 0,
|
| 263 |
+
"minerva_math500": 4
|
| 264 |
+
},
|
| 265 |
+
"higher_is_better": {
|
| 266 |
+
"gsm8k_cot_zeroshot": {
|
| 267 |
+
"exact_match": true
|
| 268 |
+
},
|
| 269 |
+
"ifeval": {
|
| 270 |
+
"prompt_level_strict_acc": true,
|
| 271 |
+
"inst_level_strict_acc": true,
|
| 272 |
+
"prompt_level_loose_acc": true,
|
| 273 |
+
"inst_level_loose_acc": true
|
| 274 |
+
},
|
| 275 |
+
"minerva_math500": {
|
| 276 |
+
"exact_match": true,
|
| 277 |
+
"math_verify": true
|
| 278 |
+
}
|
| 279 |
+
},
|
| 280 |
+
"n-samples": {
|
| 281 |
+
"gsm8k_cot_zeroshot": {
|
| 282 |
+
"original": 1319,
|
| 283 |
+
"effective": 1319
|
| 284 |
+
},
|
| 285 |
+
"minerva_math500": {
|
| 286 |
+
"original": 500,
|
| 287 |
+
"effective": 500
|
| 288 |
+
},
|
| 289 |
+
"ifeval": {
|
| 290 |
+
"original": 541,
|
| 291 |
+
"effective": 541
|
| 292 |
+
}
|
| 293 |
+
},
|
| 294 |
+
"config": {
|
| 295 |
+
"model": "local-chat-completions",
|
| 296 |
+
"model_args": {
|
| 297 |
+
"model": "student",
|
| 298 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 299 |
+
"num_concurrent": 48,
|
| 300 |
+
"tokenized_requests": false,
|
| 301 |
+
"max_retries": 3
|
| 302 |
+
},
|
| 303 |
+
"batch_size": 1,
|
| 304 |
+
"batch_sizes": [],
|
| 305 |
+
"device": "cuda:0",
|
| 306 |
+
"use_cache": null,
|
| 307 |
+
"limit": null,
|
| 308 |
+
"bootstrap_iters": 100000,
|
| 309 |
+
"gen_kwargs": {
|
| 310 |
+
"max_gen_toks": 1280
|
| 311 |
+
},
|
| 312 |
+
"random_seed": 0,
|
| 313 |
+
"numpy_seed": 1234,
|
| 314 |
+
"torch_seed": 1234,
|
| 315 |
+
"fewshot_seed": 1234
|
| 316 |
+
},
|
| 317 |
+
"git_hash": "247c7f0",
|
| 318 |
+
"date": 1784324361.384908,
|
| 319 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 320 |
+
"transformers_version": "4.57.6",
|
| 321 |
+
"lm_eval_version": "0.4.12",
|
| 322 |
+
"upper_git_hash": null,
|
| 323 |
+
"task_hashes": {
|
| 324 |
+
"gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
|
| 325 |
+
"minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
|
| 326 |
+
"ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
|
| 327 |
+
},
|
| 328 |
+
"model_source": "local-chat-completions",
|
| 329 |
+
"model_name": "student",
|
| 330 |
+
"model_name_sanitized": "student",
|
| 331 |
+
"system_instruction": null,
|
| 332 |
+
"system_instruction_sha": null,
|
| 333 |
+
"fewshot_as_multiturn": true,
|
| 334 |
+
"chat_template": "",
|
| 335 |
+
"chat_template_sha": null,
|
| 336 |
+
"total_evaluation_time_seconds": "559.1113323710015"
|
| 337 |
+
}
|
evals/general_suite/healed/glean_keep25_s1226/student/results_2026-07-17T14-54-44.918246.json
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"humaneval": {
|
| 4 |
+
"name": "humaneval",
|
| 5 |
+
"alias": "humaneval",
|
| 6 |
+
"sample_len": 164,
|
| 7 |
+
"pass@1,create_test": 0.024390243902439025,
|
| 8 |
+
"pass@1_stderr,create_test": 0.01208237572239194
|
| 9 |
+
},
|
| 10 |
+
"mbpp": {
|
| 11 |
+
"name": "mbpp",
|
| 12 |
+
"alias": "mbpp",
|
| 13 |
+
"sample_len": 500,
|
| 14 |
+
"pass_at_1,none": 0.11,
|
| 15 |
+
"pass_at_1_stderr,none": 0.014006869199415602
|
| 16 |
+
}
|
| 17 |
+
},
|
| 18 |
+
"group_subtasks": {},
|
| 19 |
+
"configs": {
|
| 20 |
+
"humaneval": {
|
| 21 |
+
"task": "humaneval",
|
| 22 |
+
"dataset_path": "openai/openai_humaneval",
|
| 23 |
+
"test_split": "test",
|
| 24 |
+
"doc_to_text": "{{prompt}}",
|
| 25 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 26 |
+
"unsafe_code": true,
|
| 27 |
+
"description": "",
|
| 28 |
+
"target_delimiter": " ",
|
| 29 |
+
"fewshot_delimiter": "\n\n",
|
| 30 |
+
"fewshot_config": {
|
| 31 |
+
"sampler": "default",
|
| 32 |
+
"split": null,
|
| 33 |
+
"process_docs": null,
|
| 34 |
+
"fewshot_indices": null,
|
| 35 |
+
"samples": null,
|
| 36 |
+
"doc_to_text": "{{prompt}}",
|
| 37 |
+
"doc_to_choice": null,
|
| 38 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 39 |
+
"gen_prefix": null,
|
| 40 |
+
"fewshot_delimiter": "\n\n",
|
| 41 |
+
"target_delimiter": " "
|
| 42 |
+
},
|
| 43 |
+
"num_fewshot": 0,
|
| 44 |
+
"metric_list": [
|
| 45 |
+
{
|
| 46 |
+
"metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
|
| 47 |
+
"aggregation": "mean",
|
| 48 |
+
"higher_is_better": true,
|
| 49 |
+
"k": [
|
| 50 |
+
1
|
| 51 |
+
]
|
| 52 |
+
}
|
| 53 |
+
],
|
| 54 |
+
"output_type": "generate_until",
|
| 55 |
+
"generation_kwargs": {
|
| 56 |
+
"until": [
|
| 57 |
+
"\nclass",
|
| 58 |
+
"\ndef",
|
| 59 |
+
"\n#",
|
| 60 |
+
"\nif",
|
| 61 |
+
"\nprint"
|
| 62 |
+
],
|
| 63 |
+
"max_gen_toks": 1024,
|
| 64 |
+
"do_sample": false
|
| 65 |
+
},
|
| 66 |
+
"repeats": 1,
|
| 67 |
+
"filter_list": [
|
| 68 |
+
{
|
| 69 |
+
"name": "create_test",
|
| 70 |
+
"filter": [
|
| 71 |
+
{
|
| 72 |
+
"function": "custom",
|
| 73 |
+
"filter_fn": "<function build_predictions at 0x789d7586aca0>"
|
| 74 |
+
}
|
| 75 |
+
]
|
| 76 |
+
}
|
| 77 |
+
],
|
| 78 |
+
"should_decontaminate": false,
|
| 79 |
+
"metadata": {
|
| 80 |
+
"version": 1.0,
|
| 81 |
+
"model": "student",
|
| 82 |
+
"base_url": "http://127.0.0.1:8422/v1/completions",
|
| 83 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep25_s1226/step0150",
|
| 84 |
+
"num_concurrent": 48,
|
| 85 |
+
"tokenized_requests": false,
|
| 86 |
+
"max_retries": 3,
|
| 87 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
|
| 88 |
+
}
|
| 89 |
+
},
|
| 90 |
+
"mbpp": {
|
| 91 |
+
"task": "mbpp",
|
| 92 |
+
"dataset_path": "google-research-datasets/mbpp",
|
| 93 |
+
"dataset_name": "full",
|
| 94 |
+
"test_split": "test",
|
| 95 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 96 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 97 |
+
"unsafe_code": true,
|
| 98 |
+
"description": "",
|
| 99 |
+
"target_delimiter": "",
|
| 100 |
+
"fewshot_delimiter": "\n\n",
|
| 101 |
+
"fewshot_config": {
|
| 102 |
+
"sampler": "first_n",
|
| 103 |
+
"split": null,
|
| 104 |
+
"process_docs": null,
|
| 105 |
+
"fewshot_indices": null,
|
| 106 |
+
"samples": "<function list_fewshot_samples at 0x789e3e90b9c0>",
|
| 107 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 108 |
+
"doc_to_choice": null,
|
| 109 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 110 |
+
"gen_prefix": null,
|
| 111 |
+
"fewshot_delimiter": "\n\n",
|
| 112 |
+
"target_delimiter": ""
|
| 113 |
+
},
|
| 114 |
+
"num_fewshot": 3,
|
| 115 |
+
"metric_list": [
|
| 116 |
+
{
|
| 117 |
+
"metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
|
| 118 |
+
"aggregation": "mean",
|
| 119 |
+
"higher_is_better": true
|
| 120 |
+
}
|
| 121 |
+
],
|
| 122 |
+
"output_type": "generate_until",
|
| 123 |
+
"generation_kwargs": {
|
| 124 |
+
"until": [
|
| 125 |
+
"[DONE]"
|
| 126 |
+
],
|
| 127 |
+
"do_sample": false
|
| 128 |
+
},
|
| 129 |
+
"repeats": 1,
|
| 130 |
+
"should_decontaminate": false,
|
| 131 |
+
"metadata": {
|
| 132 |
+
"version": 1.0,
|
| 133 |
+
"model": "student",
|
| 134 |
+
"base_url": "http://127.0.0.1:8422/v1/completions",
|
| 135 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep25_s1226/step0150",
|
| 136 |
+
"num_concurrent": 48,
|
| 137 |
+
"tokenized_requests": false,
|
| 138 |
+
"max_retries": 3,
|
| 139 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
|
| 140 |
+
}
|
| 141 |
+
}
|
| 142 |
+
},
|
| 143 |
+
"versions": {
|
| 144 |
+
"humaneval": 1.0,
|
| 145 |
+
"mbpp": 1.0
|
| 146 |
+
},
|
| 147 |
+
"n-shot": {
|
| 148 |
+
"humaneval": 0,
|
| 149 |
+
"mbpp": 3
|
| 150 |
+
},
|
| 151 |
+
"higher_is_better": {
|
| 152 |
+
"humaneval": {
|
| 153 |
+
"pass_at_k": true,
|
| 154 |
+
"pass@1": true
|
| 155 |
+
},
|
| 156 |
+
"mbpp": {
|
| 157 |
+
"pass_at_1": true
|
| 158 |
+
}
|
| 159 |
+
},
|
| 160 |
+
"n-samples": {
|
| 161 |
+
"humaneval": {
|
| 162 |
+
"original": 164,
|
| 163 |
+
"effective": 164
|
| 164 |
+
},
|
| 165 |
+
"mbpp": {
|
| 166 |
+
"original": 500,
|
| 167 |
+
"effective": 500
|
| 168 |
+
}
|
| 169 |
+
},
|
| 170 |
+
"config": {
|
| 171 |
+
"model": "local-completions",
|
| 172 |
+
"model_args": {
|
| 173 |
+
"model": "student",
|
| 174 |
+
"base_url": "http://127.0.0.1:8422/v1/completions",
|
| 175 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep25_s1226/step0150",
|
| 176 |
+
"num_concurrent": 48,
|
| 177 |
+
"tokenized_requests": false,
|
| 178 |
+
"max_retries": 3
|
| 179 |
+
},
|
| 180 |
+
"batch_size": 1,
|
| 181 |
+
"batch_sizes": [],
|
| 182 |
+
"device": "cuda:0",
|
| 183 |
+
"use_cache": null,
|
| 184 |
+
"limit": null,
|
| 185 |
+
"bootstrap_iters": 100000,
|
| 186 |
+
"gen_kwargs": {},
|
| 187 |
+
"random_seed": 0,
|
| 188 |
+
"numpy_seed": 1234,
|
| 189 |
+
"torch_seed": 1234,
|
| 190 |
+
"fewshot_seed": 1234
|
| 191 |
+
},
|
| 192 |
+
"git_hash": "247c7f0",
|
| 193 |
+
"date": 1784324923.614639,
|
| 194 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 195 |
+
"transformers_version": "4.57.6",
|
| 196 |
+
"lm_eval_version": "0.4.12",
|
| 197 |
+
"upper_git_hash": null,
|
| 198 |
+
"tokenizer_pad_token": [
|
| 199 |
+
"<pad>",
|
| 200 |
+
"50280"
|
| 201 |
+
],
|
| 202 |
+
"tokenizer_eos_token": [
|
| 203 |
+
"|||IP_ADDRESS|||",
|
| 204 |
+
"50279"
|
| 205 |
+
],
|
| 206 |
+
"tokenizer_bos_token": [
|
| 207 |
+
"|||IP_ADDRESS|||",
|
| 208 |
+
"50279"
|
| 209 |
+
],
|
| 210 |
+
"eot_token_id": 50279,
|
| 211 |
+
"max_length": 2047,
|
| 212 |
+
"task_hashes": {
|
| 213 |
+
"humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
|
| 214 |
+
"mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
|
| 215 |
+
},
|
| 216 |
+
"model_source": "local-completions",
|
| 217 |
+
"model_name": "student",
|
| 218 |
+
"model_name_sanitized": "student",
|
| 219 |
+
"system_instruction": null,
|
| 220 |
+
"system_instruction_sha": null,
|
| 221 |
+
"fewshot_as_multiturn": null,
|
| 222 |
+
"chat_template": null,
|
| 223 |
+
"chat_template_sha": null,
|
| 224 |
+
"total_evaluation_time_seconds": "367.6433539800055"
|
| 225 |
+
}
|
evals/general_suite/healed/glean_keep25_s1226/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-48-34.159099.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1226/student/samples_humaneval_2026-07-17T14-54-44.918246.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1226/student/samples_ifeval_2026-07-17T14-48-34.159099.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1226/student/samples_mbpp_2026-07-17T14-54-44.918246.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep25_s1226/student/samples_minerva_math500_2026-07-17T14-48-34.159099.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224/student/results_2026-07-18T10-11-54.989656.json
ADDED
|
@@ -0,0 +1,337 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"gsm8k_cot_zeroshot": {
|
| 4 |
+
"name": "gsm8k_cot_zeroshot",
|
| 5 |
+
"alias": "gsm8k_cot_zeroshot",
|
| 6 |
+
"sample_len": 1319,
|
| 7 |
+
"exact_match,strict-match": 0.000758150113722517,
|
| 8 |
+
"exact_match_stderr,strict-match": 0.0007581501137225239,
|
| 9 |
+
"exact_match,flexible-extract": 0.5572403335860501,
|
| 10 |
+
"exact_match_stderr,flexible-extract": 0.013681937191764623
|
| 11 |
+
},
|
| 12 |
+
"minerva_math500": {
|
| 13 |
+
"name": "minerva_math500",
|
| 14 |
+
"alias": "minerva_math500",
|
| 15 |
+
"sample_len": 500,
|
| 16 |
+
"exact_match,none": 0.118,
|
| 17 |
+
"exact_match_stderr,none": 0.014441922942480794,
|
| 18 |
+
"math_verify,none": 0.186,
|
| 19 |
+
"math_verify_stderr,none": 0.01741880678058399
|
| 20 |
+
},
|
| 21 |
+
"ifeval": {
|
| 22 |
+
"name": "ifeval",
|
| 23 |
+
"alias": "ifeval",
|
| 24 |
+
"sample_len": 541,
|
| 25 |
+
"prompt_level_strict_acc,none": 0.5582255083179297,
|
| 26 |
+
"prompt_level_strict_acc_stderr,none": 0.02137018475895099,
|
| 27 |
+
"inst_level_strict_acc,none": 0.6762589928057554,
|
| 28 |
+
"inst_level_strict_acc_stderr,none": "N/A",
|
| 29 |
+
"prompt_level_loose_acc,none": 0.5914972273567468,
|
| 30 |
+
"prompt_level_loose_acc_stderr,none": 0.021153244098720447,
|
| 31 |
+
"inst_level_loose_acc,none": 0.7086330935251799,
|
| 32 |
+
"inst_level_loose_acc_stderr,none": "N/A"
|
| 33 |
+
}
|
| 34 |
+
},
|
| 35 |
+
"group_subtasks": {},
|
| 36 |
+
"configs": {
|
| 37 |
+
"gsm8k_cot_zeroshot": {
|
| 38 |
+
"task": "gsm8k_cot_zeroshot",
|
| 39 |
+
"dataset_path": "openai/gsm8k",
|
| 40 |
+
"dataset_name": "main",
|
| 41 |
+
"training_split": "train",
|
| 42 |
+
"test_split": "test",
|
| 43 |
+
"fewshot_split": "train",
|
| 44 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 45 |
+
"doc_to_target": "{{answer}}",
|
| 46 |
+
"unsafe_code": false,
|
| 47 |
+
"description": "",
|
| 48 |
+
"target_delimiter": " ",
|
| 49 |
+
"fewshot_delimiter": "\n\n",
|
| 50 |
+
"fewshot_config": {
|
| 51 |
+
"sampler": "default",
|
| 52 |
+
"split": "train",
|
| 53 |
+
"process_docs": null,
|
| 54 |
+
"fewshot_indices": null,
|
| 55 |
+
"samples": null,
|
| 56 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 57 |
+
"doc_to_choice": null,
|
| 58 |
+
"doc_to_target": "{{answer}}",
|
| 59 |
+
"gen_prefix": null,
|
| 60 |
+
"fewshot_delimiter": "\n\n",
|
| 61 |
+
"target_delimiter": " "
|
| 62 |
+
},
|
| 63 |
+
"num_fewshot": 0,
|
| 64 |
+
"metric_list": [
|
| 65 |
+
{
|
| 66 |
+
"metric": "exact_match",
|
| 67 |
+
"aggregation": "mean",
|
| 68 |
+
"higher_is_better": true,
|
| 69 |
+
"ignore_case": true,
|
| 70 |
+
"ignore_punctuation": false,
|
| 71 |
+
"regexes_to_ignore": [
|
| 72 |
+
",",
|
| 73 |
+
"\\$",
|
| 74 |
+
"(?s).*#### ",
|
| 75 |
+
"\\.$"
|
| 76 |
+
]
|
| 77 |
+
}
|
| 78 |
+
],
|
| 79 |
+
"output_type": "generate_until",
|
| 80 |
+
"generation_kwargs": {
|
| 81 |
+
"until": [
|
| 82 |
+
"Q:",
|
| 83 |
+
"</s>",
|
| 84 |
+
"<|im_end|>"
|
| 85 |
+
],
|
| 86 |
+
"do_sample": false,
|
| 87 |
+
"max_gen_toks": 1280
|
| 88 |
+
},
|
| 89 |
+
"repeats": 1,
|
| 90 |
+
"filter_list": [
|
| 91 |
+
{
|
| 92 |
+
"name": "strict-match",
|
| 93 |
+
"filter": [
|
| 94 |
+
{
|
| 95 |
+
"function": "regex",
|
| 96 |
+
"regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"function": "take_first"
|
| 100 |
+
}
|
| 101 |
+
]
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"name": "flexible-extract",
|
| 105 |
+
"filter": [
|
| 106 |
+
{
|
| 107 |
+
"function": "regex",
|
| 108 |
+
"group_select": -1,
|
| 109 |
+
"regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"function": "take_first"
|
| 113 |
+
}
|
| 114 |
+
]
|
| 115 |
+
}
|
| 116 |
+
],
|
| 117 |
+
"should_decontaminate": false,
|
| 118 |
+
"metadata": {
|
| 119 |
+
"version": 3.0,
|
| 120 |
+
"model": "student",
|
| 121 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 122 |
+
"num_concurrent": 48,
|
| 123 |
+
"tokenized_requests": false,
|
| 124 |
+
"max_retries": 3,
|
| 125 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
|
| 126 |
+
}
|
| 127 |
+
},
|
| 128 |
+
"ifeval": {
|
| 129 |
+
"task": "ifeval",
|
| 130 |
+
"dataset_path": "google/IFEval",
|
| 131 |
+
"test_split": "train",
|
| 132 |
+
"doc_to_text": "prompt",
|
| 133 |
+
"doc_to_target": 0,
|
| 134 |
+
"unsafe_code": false,
|
| 135 |
+
"process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
|
| 136 |
+
"description": "",
|
| 137 |
+
"target_delimiter": " ",
|
| 138 |
+
"fewshot_delimiter": "\n\n",
|
| 139 |
+
"fewshot_config": {
|
| 140 |
+
"sampler": "default",
|
| 141 |
+
"split": null,
|
| 142 |
+
"process_docs": null,
|
| 143 |
+
"fewshot_indices": null,
|
| 144 |
+
"samples": null,
|
| 145 |
+
"doc_to_text": "prompt",
|
| 146 |
+
"doc_to_choice": null,
|
| 147 |
+
"doc_to_target": 0,
|
| 148 |
+
"gen_prefix": null,
|
| 149 |
+
"fewshot_delimiter": "\n\n",
|
| 150 |
+
"target_delimiter": " "
|
| 151 |
+
},
|
| 152 |
+
"num_fewshot": 0,
|
| 153 |
+
"metric_list": [
|
| 154 |
+
{
|
| 155 |
+
"metric": "prompt_level_strict_acc",
|
| 156 |
+
"aggregation": "mean",
|
| 157 |
+
"higher_is_better": true
|
| 158 |
+
},
|
| 159 |
+
{
|
| 160 |
+
"metric": "inst_level_strict_acc",
|
| 161 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 162 |
+
"higher_is_better": true
|
| 163 |
+
},
|
| 164 |
+
{
|
| 165 |
+
"metric": "prompt_level_loose_acc",
|
| 166 |
+
"aggregation": "mean",
|
| 167 |
+
"higher_is_better": true
|
| 168 |
+
},
|
| 169 |
+
{
|
| 170 |
+
"metric": "inst_level_loose_acc",
|
| 171 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 172 |
+
"higher_is_better": true
|
| 173 |
+
}
|
| 174 |
+
],
|
| 175 |
+
"output_type": "generate_until",
|
| 176 |
+
"generation_kwargs": {
|
| 177 |
+
"until": [],
|
| 178 |
+
"do_sample": false,
|
| 179 |
+
"temperature": 0.0,
|
| 180 |
+
"max_gen_toks": 1280
|
| 181 |
+
},
|
| 182 |
+
"repeats": 1,
|
| 183 |
+
"should_decontaminate": false,
|
| 184 |
+
"metadata": {
|
| 185 |
+
"version": 4.0,
|
| 186 |
+
"model": "student",
|
| 187 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 188 |
+
"num_concurrent": 48,
|
| 189 |
+
"tokenized_requests": false,
|
| 190 |
+
"max_retries": 3,
|
| 191 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
|
| 192 |
+
}
|
| 193 |
+
},
|
| 194 |
+
"minerva_math500": {
|
| 195 |
+
"task": "minerva_math500",
|
| 196 |
+
"dataset_path": "HuggingFaceH4/MATH-500",
|
| 197 |
+
"dataset_name": "default",
|
| 198 |
+
"test_split": "test",
|
| 199 |
+
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
|
| 200 |
+
"doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
|
| 201 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 202 |
+
"unsafe_code": false,
|
| 203 |
+
"process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
|
| 204 |
+
"description": "",
|
| 205 |
+
"target_delimiter": " ",
|
| 206 |
+
"fewshot_delimiter": "\n\n",
|
| 207 |
+
"fewshot_config": {
|
| 208 |
+
"sampler": "first_n",
|
| 209 |
+
"split": null,
|
| 210 |
+
"process_docs": "<function process_docs at 0x704287c25b20>",
|
| 211 |
+
"fewshot_indices": null,
|
| 212 |
+
"samples": "<function list_fewshot_samples at 0x704287c279c0>",
|
| 213 |
+
"doc_to_text": "<function doc_to_text at 0x70428d3f18a0>",
|
| 214 |
+
"doc_to_choice": null,
|
| 215 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 216 |
+
"gen_prefix": null,
|
| 217 |
+
"fewshot_delimiter": "\n\n",
|
| 218 |
+
"target_delimiter": " "
|
| 219 |
+
},
|
| 220 |
+
"num_fewshot": 4,
|
| 221 |
+
"metric_list": [
|
| 222 |
+
{
|
| 223 |
+
"metric": "exact_match",
|
| 224 |
+
"aggregation": "mean",
|
| 225 |
+
"higher_is_better": true
|
| 226 |
+
},
|
| 227 |
+
{
|
| 228 |
+
"metric": "math_verify",
|
| 229 |
+
"aggregation": "mean",
|
| 230 |
+
"higher_is_better": true
|
| 231 |
+
}
|
| 232 |
+
],
|
| 233 |
+
"output_type": "generate_until",
|
| 234 |
+
"generation_kwargs": {
|
| 235 |
+
"until": [
|
| 236 |
+
"Problem:"
|
| 237 |
+
],
|
| 238 |
+
"do_sample": false,
|
| 239 |
+
"temperature": 0.0,
|
| 240 |
+
"max_gen_toks": 1280
|
| 241 |
+
},
|
| 242 |
+
"repeats": 1,
|
| 243 |
+
"should_decontaminate": false,
|
| 244 |
+
"metadata": {
|
| 245 |
+
"version": 3.0,
|
| 246 |
+
"model": "student",
|
| 247 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 248 |
+
"num_concurrent": 48,
|
| 249 |
+
"tokenized_requests": false,
|
| 250 |
+
"max_retries": 3,
|
| 251 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
|
| 252 |
+
}
|
| 253 |
+
}
|
| 254 |
+
},
|
| 255 |
+
"versions": {
|
| 256 |
+
"gsm8k_cot_zeroshot": 3.0,
|
| 257 |
+
"ifeval": 4.0,
|
| 258 |
+
"minerva_math500": 3.0
|
| 259 |
+
},
|
| 260 |
+
"n-shot": {
|
| 261 |
+
"gsm8k_cot_zeroshot": 0,
|
| 262 |
+
"ifeval": 0,
|
| 263 |
+
"minerva_math500": 4
|
| 264 |
+
},
|
| 265 |
+
"higher_is_better": {
|
| 266 |
+
"gsm8k_cot_zeroshot": {
|
| 267 |
+
"exact_match": true
|
| 268 |
+
},
|
| 269 |
+
"ifeval": {
|
| 270 |
+
"prompt_level_strict_acc": true,
|
| 271 |
+
"inst_level_strict_acc": true,
|
| 272 |
+
"prompt_level_loose_acc": true,
|
| 273 |
+
"inst_level_loose_acc": true
|
| 274 |
+
},
|
| 275 |
+
"minerva_math500": {
|
| 276 |
+
"exact_match": true,
|
| 277 |
+
"math_verify": true
|
| 278 |
+
}
|
| 279 |
+
},
|
| 280 |
+
"n-samples": {
|
| 281 |
+
"gsm8k_cot_zeroshot": {
|
| 282 |
+
"original": 1319,
|
| 283 |
+
"effective": 1319
|
| 284 |
+
},
|
| 285 |
+
"minerva_math500": {
|
| 286 |
+
"original": 500,
|
| 287 |
+
"effective": 500
|
| 288 |
+
},
|
| 289 |
+
"ifeval": {
|
| 290 |
+
"original": 541,
|
| 291 |
+
"effective": 541
|
| 292 |
+
}
|
| 293 |
+
},
|
| 294 |
+
"config": {
|
| 295 |
+
"model": "local-chat-completions",
|
| 296 |
+
"model_args": {
|
| 297 |
+
"model": "student",
|
| 298 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 299 |
+
"num_concurrent": 48,
|
| 300 |
+
"tokenized_requests": false,
|
| 301 |
+
"max_retries": 3
|
| 302 |
+
},
|
| 303 |
+
"batch_size": 1,
|
| 304 |
+
"batch_sizes": [],
|
| 305 |
+
"device": "cuda:0",
|
| 306 |
+
"use_cache": null,
|
| 307 |
+
"limit": null,
|
| 308 |
+
"bootstrap_iters": 100000,
|
| 309 |
+
"gen_kwargs": {
|
| 310 |
+
"max_gen_toks": 1280
|
| 311 |
+
},
|
| 312 |
+
"random_seed": 0,
|
| 313 |
+
"numpy_seed": 1234,
|
| 314 |
+
"torch_seed": 1234,
|
| 315 |
+
"fewshot_seed": 1234
|
| 316 |
+
},
|
| 317 |
+
"git_hash": "4cd474a",
|
| 318 |
+
"date": 1784394159.0160384,
|
| 319 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 320 |
+
"transformers_version": "4.57.6",
|
| 321 |
+
"lm_eval_version": "0.4.12",
|
| 322 |
+
"upper_git_hash": null,
|
| 323 |
+
"task_hashes": {
|
| 324 |
+
"gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
|
| 325 |
+
"minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
|
| 326 |
+
"ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
|
| 327 |
+
},
|
| 328 |
+
"model_source": "local-chat-completions",
|
| 329 |
+
"model_name": "student",
|
| 330 |
+
"model_name_sanitized": "student",
|
| 331 |
+
"system_instruction": null,
|
| 332 |
+
"system_instruction_sha": null,
|
| 333 |
+
"fewshot_as_multiturn": true,
|
| 334 |
+
"chat_template": "",
|
| 335 |
+
"chat_template_sha": null,
|
| 336 |
+
"total_evaluation_time_seconds": "562.3516382400012"
|
| 337 |
+
}
|
evals/general_suite/healed/glean_keep50_s1224/student/results_2026-07-18T10-15-21.380553.json
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"humaneval": {
|
| 4 |
+
"name": "humaneval",
|
| 5 |
+
"alias": "humaneval",
|
| 6 |
+
"sample_len": 164,
|
| 7 |
+
"pass@1,create_test": 0.2682926829268293,
|
| 8 |
+
"pass@1_stderr,create_test": 0.03470398212814533
|
| 9 |
+
},
|
| 10 |
+
"mbpp": {
|
| 11 |
+
"name": "mbpp",
|
| 12 |
+
"alias": "mbpp",
|
| 13 |
+
"sample_len": 500,
|
| 14 |
+
"pass_at_1,none": 0.23,
|
| 15 |
+
"pass_at_1_stderr,none": 0.018839050391123137
|
| 16 |
+
}
|
| 17 |
+
},
|
| 18 |
+
"group_subtasks": {},
|
| 19 |
+
"configs": {
|
| 20 |
+
"humaneval": {
|
| 21 |
+
"task": "humaneval",
|
| 22 |
+
"dataset_path": "openai/openai_humaneval",
|
| 23 |
+
"test_split": "test",
|
| 24 |
+
"doc_to_text": "{{prompt}}",
|
| 25 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 26 |
+
"unsafe_code": true,
|
| 27 |
+
"description": "",
|
| 28 |
+
"target_delimiter": " ",
|
| 29 |
+
"fewshot_delimiter": "\n\n",
|
| 30 |
+
"fewshot_config": {
|
| 31 |
+
"sampler": "default",
|
| 32 |
+
"split": null,
|
| 33 |
+
"process_docs": null,
|
| 34 |
+
"fewshot_indices": null,
|
| 35 |
+
"samples": null,
|
| 36 |
+
"doc_to_text": "{{prompt}}",
|
| 37 |
+
"doc_to_choice": null,
|
| 38 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 39 |
+
"gen_prefix": null,
|
| 40 |
+
"fewshot_delimiter": "\n\n",
|
| 41 |
+
"target_delimiter": " "
|
| 42 |
+
},
|
| 43 |
+
"num_fewshot": 0,
|
| 44 |
+
"metric_list": [
|
| 45 |
+
{
|
| 46 |
+
"metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
|
| 47 |
+
"aggregation": "mean",
|
| 48 |
+
"higher_is_better": true,
|
| 49 |
+
"k": [
|
| 50 |
+
1
|
| 51 |
+
]
|
| 52 |
+
}
|
| 53 |
+
],
|
| 54 |
+
"output_type": "generate_until",
|
| 55 |
+
"generation_kwargs": {
|
| 56 |
+
"until": [
|
| 57 |
+
"\nclass",
|
| 58 |
+
"\ndef",
|
| 59 |
+
"\n#",
|
| 60 |
+
"\nif",
|
| 61 |
+
"\nprint"
|
| 62 |
+
],
|
| 63 |
+
"max_gen_toks": 1024,
|
| 64 |
+
"do_sample": false
|
| 65 |
+
},
|
| 66 |
+
"repeats": 1,
|
| 67 |
+
"filter_list": [
|
| 68 |
+
{
|
| 69 |
+
"name": "create_test",
|
| 70 |
+
"filter": [
|
| 71 |
+
{
|
| 72 |
+
"function": "custom",
|
| 73 |
+
"filter_fn": "<function build_predictions at 0x7d20f9872ca0>"
|
| 74 |
+
}
|
| 75 |
+
]
|
| 76 |
+
}
|
| 77 |
+
],
|
| 78 |
+
"should_decontaminate": false,
|
| 79 |
+
"metadata": {
|
| 80 |
+
"version": 1.0,
|
| 81 |
+
"model": "student",
|
| 82 |
+
"base_url": "http://127.0.0.1:8420/v1/completions",
|
| 83 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0150",
|
| 84 |
+
"num_concurrent": 48,
|
| 85 |
+
"tokenized_requests": false,
|
| 86 |
+
"max_retries": 3,
|
| 87 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
|
| 88 |
+
}
|
| 89 |
+
},
|
| 90 |
+
"mbpp": {
|
| 91 |
+
"task": "mbpp",
|
| 92 |
+
"dataset_path": "google-research-datasets/mbpp",
|
| 93 |
+
"dataset_name": "full",
|
| 94 |
+
"test_split": "test",
|
| 95 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 96 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 97 |
+
"unsafe_code": true,
|
| 98 |
+
"description": "",
|
| 99 |
+
"target_delimiter": "",
|
| 100 |
+
"fewshot_delimiter": "\n\n",
|
| 101 |
+
"fewshot_config": {
|
| 102 |
+
"sampler": "first_n",
|
| 103 |
+
"split": null,
|
| 104 |
+
"process_docs": null,
|
| 105 |
+
"fewshot_indices": null,
|
| 106 |
+
"samples": "<function list_fewshot_samples at 0x7d21c2b1f9c0>",
|
| 107 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 108 |
+
"doc_to_choice": null,
|
| 109 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 110 |
+
"gen_prefix": null,
|
| 111 |
+
"fewshot_delimiter": "\n\n",
|
| 112 |
+
"target_delimiter": ""
|
| 113 |
+
},
|
| 114 |
+
"num_fewshot": 3,
|
| 115 |
+
"metric_list": [
|
| 116 |
+
{
|
| 117 |
+
"metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
|
| 118 |
+
"aggregation": "mean",
|
| 119 |
+
"higher_is_better": true
|
| 120 |
+
}
|
| 121 |
+
],
|
| 122 |
+
"output_type": "generate_until",
|
| 123 |
+
"generation_kwargs": {
|
| 124 |
+
"until": [
|
| 125 |
+
"[DONE]"
|
| 126 |
+
],
|
| 127 |
+
"do_sample": false
|
| 128 |
+
},
|
| 129 |
+
"repeats": 1,
|
| 130 |
+
"should_decontaminate": false,
|
| 131 |
+
"metadata": {
|
| 132 |
+
"version": 1.0,
|
| 133 |
+
"model": "student",
|
| 134 |
+
"base_url": "http://127.0.0.1:8420/v1/completions",
|
| 135 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0150",
|
| 136 |
+
"num_concurrent": 48,
|
| 137 |
+
"tokenized_requests": false,
|
| 138 |
+
"max_retries": 3,
|
| 139 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
|
| 140 |
+
}
|
| 141 |
+
}
|
| 142 |
+
},
|
| 143 |
+
"versions": {
|
| 144 |
+
"humaneval": 1.0,
|
| 145 |
+
"mbpp": 1.0
|
| 146 |
+
},
|
| 147 |
+
"n-shot": {
|
| 148 |
+
"humaneval": 0,
|
| 149 |
+
"mbpp": 3
|
| 150 |
+
},
|
| 151 |
+
"higher_is_better": {
|
| 152 |
+
"humaneval": {
|
| 153 |
+
"pass_at_k": true,
|
| 154 |
+
"pass@1": true
|
| 155 |
+
},
|
| 156 |
+
"mbpp": {
|
| 157 |
+
"pass_at_1": true
|
| 158 |
+
}
|
| 159 |
+
},
|
| 160 |
+
"n-samples": {
|
| 161 |
+
"humaneval": {
|
| 162 |
+
"original": 164,
|
| 163 |
+
"effective": 164
|
| 164 |
+
},
|
| 165 |
+
"mbpp": {
|
| 166 |
+
"original": 500,
|
| 167 |
+
"effective": 500
|
| 168 |
+
}
|
| 169 |
+
},
|
| 170 |
+
"config": {
|
| 171 |
+
"model": "local-completions",
|
| 172 |
+
"model_args": {
|
| 173 |
+
"model": "student",
|
| 174 |
+
"base_url": "http://127.0.0.1:8420/v1/completions",
|
| 175 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0150",
|
| 176 |
+
"num_concurrent": 48,
|
| 177 |
+
"tokenized_requests": false,
|
| 178 |
+
"max_retries": 3
|
| 179 |
+
},
|
| 180 |
+
"batch_size": 1,
|
| 181 |
+
"batch_sizes": [],
|
| 182 |
+
"device": "cuda:0",
|
| 183 |
+
"use_cache": null,
|
| 184 |
+
"limit": null,
|
| 185 |
+
"bootstrap_iters": 100000,
|
| 186 |
+
"gen_kwargs": {},
|
| 187 |
+
"random_seed": 0,
|
| 188 |
+
"numpy_seed": 1234,
|
| 189 |
+
"torch_seed": 1234,
|
| 190 |
+
"fewshot_seed": 1234
|
| 191 |
+
},
|
| 192 |
+
"git_hash": "4cd474a",
|
| 193 |
+
"date": 1784394724.0320532,
|
| 194 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 195 |
+
"transformers_version": "4.57.6",
|
| 196 |
+
"lm_eval_version": "0.4.12",
|
| 197 |
+
"upper_git_hash": null,
|
| 198 |
+
"tokenizer_pad_token": [
|
| 199 |
+
"<pad>",
|
| 200 |
+
"50280"
|
| 201 |
+
],
|
| 202 |
+
"tokenizer_eos_token": [
|
| 203 |
+
"|||IP_ADDRESS|||",
|
| 204 |
+
"50279"
|
| 205 |
+
],
|
| 206 |
+
"tokenizer_bos_token": [
|
| 207 |
+
"|||IP_ADDRESS|||",
|
| 208 |
+
"50279"
|
| 209 |
+
],
|
| 210 |
+
"eot_token_id": 50279,
|
| 211 |
+
"max_length": 2047,
|
| 212 |
+
"task_hashes": {
|
| 213 |
+
"humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
|
| 214 |
+
"mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
|
| 215 |
+
},
|
| 216 |
+
"model_source": "local-completions",
|
| 217 |
+
"model_name": "student",
|
| 218 |
+
"model_name_sanitized": "student",
|
| 219 |
+
"system_instruction": null,
|
| 220 |
+
"system_instruction_sha": null,
|
| 221 |
+
"fewshot_as_multiturn": null,
|
| 222 |
+
"chat_template": null,
|
| 223 |
+
"chat_template_sha": null,
|
| 224 |
+
"total_evaluation_time_seconds": "203.6669814880006"
|
| 225 |
+
}
|
evals/general_suite/healed/glean_keep50_s1224/student/samples_gsm8k_cot_zeroshot_2026-07-18T10-11-54.989656.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224/student/samples_humaneval_2026-07-18T10-15-21.380553.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224/student/samples_ifeval_2026-07-18T10-11-54.989656.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224/student/samples_mbpp_2026-07-18T10-15-21.380553.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224/student/samples_minerva_math500_2026-07-18T10-11-54.989656.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/results_2026-07-19T21-36-45.489848.json
ADDED
|
@@ -0,0 +1,337 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"gsm8k_cot_zeroshot": {
|
| 4 |
+
"name": "gsm8k_cot_zeroshot",
|
| 5 |
+
"alias": "gsm8k_cot_zeroshot",
|
| 6 |
+
"sample_len": 1319,
|
| 7 |
+
"exact_match,strict-match": 0.0,
|
| 8 |
+
"exact_match_stderr,strict-match": 0.0,
|
| 9 |
+
"exact_match,flexible-extract": 0.5526914329037149,
|
| 10 |
+
"exact_match_stderr,flexible-extract": 0.013695795709089898
|
| 11 |
+
},
|
| 12 |
+
"minerva_math500": {
|
| 13 |
+
"name": "minerva_math500",
|
| 14 |
+
"alias": "minerva_math500",
|
| 15 |
+
"sample_len": 500,
|
| 16 |
+
"exact_match,none": 0.118,
|
| 17 |
+
"exact_match_stderr,none": 0.014441922942480794,
|
| 18 |
+
"math_verify,none": 0.21,
|
| 19 |
+
"math_verify_stderr,none": 0.01823362086530592
|
| 20 |
+
},
|
| 21 |
+
"ifeval": {
|
| 22 |
+
"name": "ifeval",
|
| 23 |
+
"alias": "ifeval",
|
| 24 |
+
"sample_len": 541,
|
| 25 |
+
"prompt_level_strict_acc,none": 0.5415896487985212,
|
| 26 |
+
"prompt_level_strict_acc_stderr,none": 0.021442010560476468,
|
| 27 |
+
"inst_level_strict_acc,none": 0.6630695443645084,
|
| 28 |
+
"inst_level_strict_acc_stderr,none": "N/A",
|
| 29 |
+
"prompt_level_loose_acc,none": 0.5656192236598891,
|
| 30 |
+
"prompt_level_loose_acc_stderr,none": 0.021330473657564707,
|
| 31 |
+
"inst_level_loose_acc,none": 0.6906474820143885,
|
| 32 |
+
"inst_level_loose_acc_stderr,none": "N/A"
|
| 33 |
+
}
|
| 34 |
+
},
|
| 35 |
+
"group_subtasks": {},
|
| 36 |
+
"configs": {
|
| 37 |
+
"gsm8k_cot_zeroshot": {
|
| 38 |
+
"task": "gsm8k_cot_zeroshot",
|
| 39 |
+
"dataset_path": "openai/gsm8k",
|
| 40 |
+
"dataset_name": "main",
|
| 41 |
+
"training_split": "train",
|
| 42 |
+
"test_split": "test",
|
| 43 |
+
"fewshot_split": "train",
|
| 44 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 45 |
+
"doc_to_target": "{{answer}}",
|
| 46 |
+
"unsafe_code": false,
|
| 47 |
+
"description": "",
|
| 48 |
+
"target_delimiter": " ",
|
| 49 |
+
"fewshot_delimiter": "\n\n",
|
| 50 |
+
"fewshot_config": {
|
| 51 |
+
"sampler": "default",
|
| 52 |
+
"split": "train",
|
| 53 |
+
"process_docs": null,
|
| 54 |
+
"fewshot_indices": null,
|
| 55 |
+
"samples": null,
|
| 56 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 57 |
+
"doc_to_choice": null,
|
| 58 |
+
"doc_to_target": "{{answer}}",
|
| 59 |
+
"gen_prefix": null,
|
| 60 |
+
"fewshot_delimiter": "\n\n",
|
| 61 |
+
"target_delimiter": " "
|
| 62 |
+
},
|
| 63 |
+
"num_fewshot": 0,
|
| 64 |
+
"metric_list": [
|
| 65 |
+
{
|
| 66 |
+
"metric": "exact_match",
|
| 67 |
+
"aggregation": "mean",
|
| 68 |
+
"higher_is_better": true,
|
| 69 |
+
"ignore_case": true,
|
| 70 |
+
"ignore_punctuation": false,
|
| 71 |
+
"regexes_to_ignore": [
|
| 72 |
+
",",
|
| 73 |
+
"\\$",
|
| 74 |
+
"(?s).*#### ",
|
| 75 |
+
"\\.$"
|
| 76 |
+
]
|
| 77 |
+
}
|
| 78 |
+
],
|
| 79 |
+
"output_type": "generate_until",
|
| 80 |
+
"generation_kwargs": {
|
| 81 |
+
"until": [
|
| 82 |
+
"Q:",
|
| 83 |
+
"</s>",
|
| 84 |
+
"<|im_end|>"
|
| 85 |
+
],
|
| 86 |
+
"do_sample": false,
|
| 87 |
+
"max_gen_toks": 1280
|
| 88 |
+
},
|
| 89 |
+
"repeats": 1,
|
| 90 |
+
"filter_list": [
|
| 91 |
+
{
|
| 92 |
+
"name": "strict-match",
|
| 93 |
+
"filter": [
|
| 94 |
+
{
|
| 95 |
+
"function": "regex",
|
| 96 |
+
"regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"function": "take_first"
|
| 100 |
+
}
|
| 101 |
+
]
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"name": "flexible-extract",
|
| 105 |
+
"filter": [
|
| 106 |
+
{
|
| 107 |
+
"function": "regex",
|
| 108 |
+
"group_select": -1,
|
| 109 |
+
"regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"function": "take_first"
|
| 113 |
+
}
|
| 114 |
+
]
|
| 115 |
+
}
|
| 116 |
+
],
|
| 117 |
+
"should_decontaminate": false,
|
| 118 |
+
"metadata": {
|
| 119 |
+
"version": 3.0,
|
| 120 |
+
"model": "student",
|
| 121 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 122 |
+
"num_concurrent": 48,
|
| 123 |
+
"tokenized_requests": false,
|
| 124 |
+
"max_retries": 3,
|
| 125 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
|
| 126 |
+
}
|
| 127 |
+
},
|
| 128 |
+
"ifeval": {
|
| 129 |
+
"task": "ifeval",
|
| 130 |
+
"dataset_path": "google/IFEval",
|
| 131 |
+
"test_split": "train",
|
| 132 |
+
"doc_to_text": "prompt",
|
| 133 |
+
"doc_to_target": 0,
|
| 134 |
+
"unsafe_code": false,
|
| 135 |
+
"process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
|
| 136 |
+
"description": "",
|
| 137 |
+
"target_delimiter": " ",
|
| 138 |
+
"fewshot_delimiter": "\n\n",
|
| 139 |
+
"fewshot_config": {
|
| 140 |
+
"sampler": "default",
|
| 141 |
+
"split": null,
|
| 142 |
+
"process_docs": null,
|
| 143 |
+
"fewshot_indices": null,
|
| 144 |
+
"samples": null,
|
| 145 |
+
"doc_to_text": "prompt",
|
| 146 |
+
"doc_to_choice": null,
|
| 147 |
+
"doc_to_target": 0,
|
| 148 |
+
"gen_prefix": null,
|
| 149 |
+
"fewshot_delimiter": "\n\n",
|
| 150 |
+
"target_delimiter": " "
|
| 151 |
+
},
|
| 152 |
+
"num_fewshot": 0,
|
| 153 |
+
"metric_list": [
|
| 154 |
+
{
|
| 155 |
+
"metric": "prompt_level_strict_acc",
|
| 156 |
+
"aggregation": "mean",
|
| 157 |
+
"higher_is_better": true
|
| 158 |
+
},
|
| 159 |
+
{
|
| 160 |
+
"metric": "inst_level_strict_acc",
|
| 161 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 162 |
+
"higher_is_better": true
|
| 163 |
+
},
|
| 164 |
+
{
|
| 165 |
+
"metric": "prompt_level_loose_acc",
|
| 166 |
+
"aggregation": "mean",
|
| 167 |
+
"higher_is_better": true
|
| 168 |
+
},
|
| 169 |
+
{
|
| 170 |
+
"metric": "inst_level_loose_acc",
|
| 171 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 172 |
+
"higher_is_better": true
|
| 173 |
+
}
|
| 174 |
+
],
|
| 175 |
+
"output_type": "generate_until",
|
| 176 |
+
"generation_kwargs": {
|
| 177 |
+
"until": [],
|
| 178 |
+
"do_sample": false,
|
| 179 |
+
"temperature": 0.0,
|
| 180 |
+
"max_gen_toks": 1280
|
| 181 |
+
},
|
| 182 |
+
"repeats": 1,
|
| 183 |
+
"should_decontaminate": false,
|
| 184 |
+
"metadata": {
|
| 185 |
+
"version": 4.0,
|
| 186 |
+
"model": "student",
|
| 187 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 188 |
+
"num_concurrent": 48,
|
| 189 |
+
"tokenized_requests": false,
|
| 190 |
+
"max_retries": 3,
|
| 191 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
|
| 192 |
+
}
|
| 193 |
+
},
|
| 194 |
+
"minerva_math500": {
|
| 195 |
+
"task": "minerva_math500",
|
| 196 |
+
"dataset_path": "HuggingFaceH4/MATH-500",
|
| 197 |
+
"dataset_name": "default",
|
| 198 |
+
"test_split": "test",
|
| 199 |
+
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
|
| 200 |
+
"doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
|
| 201 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 202 |
+
"unsafe_code": false,
|
| 203 |
+
"process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
|
| 204 |
+
"description": "",
|
| 205 |
+
"target_delimiter": " ",
|
| 206 |
+
"fewshot_delimiter": "\n\n",
|
| 207 |
+
"fewshot_config": {
|
| 208 |
+
"sampler": "first_n",
|
| 209 |
+
"split": null,
|
| 210 |
+
"process_docs": "<function process_docs at 0x74c074c25b20>",
|
| 211 |
+
"fewshot_indices": null,
|
| 212 |
+
"samples": "<function list_fewshot_samples at 0x74c074c279c0>",
|
| 213 |
+
"doc_to_text": "<function doc_to_text at 0x74c0763f1800>",
|
| 214 |
+
"doc_to_choice": null,
|
| 215 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 216 |
+
"gen_prefix": null,
|
| 217 |
+
"fewshot_delimiter": "\n\n",
|
| 218 |
+
"target_delimiter": " "
|
| 219 |
+
},
|
| 220 |
+
"num_fewshot": 4,
|
| 221 |
+
"metric_list": [
|
| 222 |
+
{
|
| 223 |
+
"metric": "exact_match",
|
| 224 |
+
"aggregation": "mean",
|
| 225 |
+
"higher_is_better": true
|
| 226 |
+
},
|
| 227 |
+
{
|
| 228 |
+
"metric": "math_verify",
|
| 229 |
+
"aggregation": "mean",
|
| 230 |
+
"higher_is_better": true
|
| 231 |
+
}
|
| 232 |
+
],
|
| 233 |
+
"output_type": "generate_until",
|
| 234 |
+
"generation_kwargs": {
|
| 235 |
+
"until": [
|
| 236 |
+
"Problem:"
|
| 237 |
+
],
|
| 238 |
+
"do_sample": false,
|
| 239 |
+
"temperature": 0.0,
|
| 240 |
+
"max_gen_toks": 1280
|
| 241 |
+
},
|
| 242 |
+
"repeats": 1,
|
| 243 |
+
"should_decontaminate": false,
|
| 244 |
+
"metadata": {
|
| 245 |
+
"version": 3.0,
|
| 246 |
+
"model": "student",
|
| 247 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 248 |
+
"num_concurrent": 48,
|
| 249 |
+
"tokenized_requests": false,
|
| 250 |
+
"max_retries": 3,
|
| 251 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
|
| 252 |
+
}
|
| 253 |
+
}
|
| 254 |
+
},
|
| 255 |
+
"versions": {
|
| 256 |
+
"gsm8k_cot_zeroshot": 3.0,
|
| 257 |
+
"ifeval": 4.0,
|
| 258 |
+
"minerva_math500": 3.0
|
| 259 |
+
},
|
| 260 |
+
"n-shot": {
|
| 261 |
+
"gsm8k_cot_zeroshot": 0,
|
| 262 |
+
"ifeval": 0,
|
| 263 |
+
"minerva_math500": 4
|
| 264 |
+
},
|
| 265 |
+
"higher_is_better": {
|
| 266 |
+
"gsm8k_cot_zeroshot": {
|
| 267 |
+
"exact_match": true
|
| 268 |
+
},
|
| 269 |
+
"ifeval": {
|
| 270 |
+
"prompt_level_strict_acc": true,
|
| 271 |
+
"inst_level_strict_acc": true,
|
| 272 |
+
"prompt_level_loose_acc": true,
|
| 273 |
+
"inst_level_loose_acc": true
|
| 274 |
+
},
|
| 275 |
+
"minerva_math500": {
|
| 276 |
+
"exact_match": true,
|
| 277 |
+
"math_verify": true
|
| 278 |
+
}
|
| 279 |
+
},
|
| 280 |
+
"n-samples": {
|
| 281 |
+
"gsm8k_cot_zeroshot": {
|
| 282 |
+
"original": 1319,
|
| 283 |
+
"effective": 1319
|
| 284 |
+
},
|
| 285 |
+
"minerva_math500": {
|
| 286 |
+
"original": 500,
|
| 287 |
+
"effective": 500
|
| 288 |
+
},
|
| 289 |
+
"ifeval": {
|
| 290 |
+
"original": 541,
|
| 291 |
+
"effective": 541
|
| 292 |
+
}
|
| 293 |
+
},
|
| 294 |
+
"config": {
|
| 295 |
+
"model": "local-chat-completions",
|
| 296 |
+
"model_args": {
|
| 297 |
+
"model": "student",
|
| 298 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 299 |
+
"num_concurrent": 48,
|
| 300 |
+
"tokenized_requests": false,
|
| 301 |
+
"max_retries": 3
|
| 302 |
+
},
|
| 303 |
+
"batch_size": 1,
|
| 304 |
+
"batch_sizes": [],
|
| 305 |
+
"device": "cuda:0",
|
| 306 |
+
"use_cache": null,
|
| 307 |
+
"limit": null,
|
| 308 |
+
"bootstrap_iters": 100000,
|
| 309 |
+
"gen_kwargs": {
|
| 310 |
+
"max_gen_toks": 1280
|
| 311 |
+
},
|
| 312 |
+
"random_seed": 0,
|
| 313 |
+
"numpy_seed": 1234,
|
| 314 |
+
"torch_seed": 1234,
|
| 315 |
+
"fewshot_seed": 1234
|
| 316 |
+
},
|
| 317 |
+
"git_hash": "4cd474a",
|
| 318 |
+
"date": 1784521647.9592357,
|
| 319 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 320 |
+
"transformers_version": "4.57.6",
|
| 321 |
+
"lm_eval_version": "0.4.12",
|
| 322 |
+
"upper_git_hash": null,
|
| 323 |
+
"task_hashes": {
|
| 324 |
+
"gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
|
| 325 |
+
"minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
|
| 326 |
+
"ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
|
| 327 |
+
},
|
| 328 |
+
"model_source": "local-chat-completions",
|
| 329 |
+
"model_name": "student",
|
| 330 |
+
"model_name_sanitized": "student",
|
| 331 |
+
"system_instruction": null,
|
| 332 |
+
"system_instruction_sha": null,
|
| 333 |
+
"fewshot_as_multiturn": true,
|
| 334 |
+
"chat_template": "",
|
| 335 |
+
"chat_template_sha": null,
|
| 336 |
+
"total_evaluation_time_seconds": "563.8881401320104"
|
| 337 |
+
}
|
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/results_2026-07-19T21-40-17.349569.json
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"humaneval": {
|
| 4 |
+
"name": "humaneval",
|
| 5 |
+
"alias": "humaneval",
|
| 6 |
+
"sample_len": 164,
|
| 7 |
+
"pass@1,create_test": 0.2804878048780488,
|
| 8 |
+
"pass@1_stderr,create_test": 0.0351870022880158
|
| 9 |
+
},
|
| 10 |
+
"mbpp": {
|
| 11 |
+
"name": "mbpp",
|
| 12 |
+
"alias": "mbpp",
|
| 13 |
+
"sample_len": 500,
|
| 14 |
+
"pass_at_1,none": 0.226,
|
| 15 |
+
"pass_at_1_stderr,none": 0.018722956449139915
|
| 16 |
+
}
|
| 17 |
+
},
|
| 18 |
+
"group_subtasks": {},
|
| 19 |
+
"configs": {
|
| 20 |
+
"humaneval": {
|
| 21 |
+
"task": "humaneval",
|
| 22 |
+
"dataset_path": "openai/openai_humaneval",
|
| 23 |
+
"test_split": "test",
|
| 24 |
+
"doc_to_text": "{{prompt}}",
|
| 25 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 26 |
+
"unsafe_code": true,
|
| 27 |
+
"description": "",
|
| 28 |
+
"target_delimiter": " ",
|
| 29 |
+
"fewshot_delimiter": "\n\n",
|
| 30 |
+
"fewshot_config": {
|
| 31 |
+
"sampler": "default",
|
| 32 |
+
"split": null,
|
| 33 |
+
"process_docs": null,
|
| 34 |
+
"fewshot_indices": null,
|
| 35 |
+
"samples": null,
|
| 36 |
+
"doc_to_text": "{{prompt}}",
|
| 37 |
+
"doc_to_choice": null,
|
| 38 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 39 |
+
"gen_prefix": null,
|
| 40 |
+
"fewshot_delimiter": "\n\n",
|
| 41 |
+
"target_delimiter": " "
|
| 42 |
+
},
|
| 43 |
+
"num_fewshot": 0,
|
| 44 |
+
"metric_list": [
|
| 45 |
+
{
|
| 46 |
+
"metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
|
| 47 |
+
"aggregation": "mean",
|
| 48 |
+
"higher_is_better": true,
|
| 49 |
+
"k": [
|
| 50 |
+
1
|
| 51 |
+
]
|
| 52 |
+
}
|
| 53 |
+
],
|
| 54 |
+
"output_type": "generate_until",
|
| 55 |
+
"generation_kwargs": {
|
| 56 |
+
"until": [
|
| 57 |
+
"\nclass",
|
| 58 |
+
"\ndef",
|
| 59 |
+
"\n#",
|
| 60 |
+
"\nif",
|
| 61 |
+
"\nprint"
|
| 62 |
+
],
|
| 63 |
+
"max_gen_toks": 1024,
|
| 64 |
+
"do_sample": false
|
| 65 |
+
},
|
| 66 |
+
"repeats": 1,
|
| 67 |
+
"filter_list": [
|
| 68 |
+
{
|
| 69 |
+
"name": "create_test",
|
| 70 |
+
"filter": [
|
| 71 |
+
{
|
| 72 |
+
"function": "custom",
|
| 73 |
+
"filter_fn": "<function build_predictions at 0x767d37f56ca0>"
|
| 74 |
+
}
|
| 75 |
+
]
|
| 76 |
+
}
|
| 77 |
+
],
|
| 78 |
+
"should_decontaminate": false,
|
| 79 |
+
"metadata": {
|
| 80 |
+
"version": 1.0,
|
| 81 |
+
"model": "student",
|
| 82 |
+
"base_url": "http://127.0.0.1:8422/v1/completions",
|
| 83 |
+
"tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0150",
|
| 84 |
+
"num_concurrent": 48,
|
| 85 |
+
"tokenized_requests": false,
|
| 86 |
+
"max_retries": 3,
|
| 87 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
|
| 88 |
+
}
|
| 89 |
+
},
|
| 90 |
+
"mbpp": {
|
| 91 |
+
"task": "mbpp",
|
| 92 |
+
"dataset_path": "google-research-datasets/mbpp",
|
| 93 |
+
"dataset_name": "full",
|
| 94 |
+
"test_split": "test",
|
| 95 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 96 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 97 |
+
"unsafe_code": true,
|
| 98 |
+
"description": "",
|
| 99 |
+
"target_delimiter": "",
|
| 100 |
+
"fewshot_delimiter": "\n\n",
|
| 101 |
+
"fewshot_config": {
|
| 102 |
+
"sampler": "first_n",
|
| 103 |
+
"split": null,
|
| 104 |
+
"process_docs": null,
|
| 105 |
+
"fewshot_indices": null,
|
| 106 |
+
"samples": "<function list_fewshot_samples at 0x767e00cef9c0>",
|
| 107 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 108 |
+
"doc_to_choice": null,
|
| 109 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 110 |
+
"gen_prefix": null,
|
| 111 |
+
"fewshot_delimiter": "\n\n",
|
| 112 |
+
"target_delimiter": ""
|
| 113 |
+
},
|
| 114 |
+
"num_fewshot": 3,
|
| 115 |
+
"metric_list": [
|
| 116 |
+
{
|
| 117 |
+
"metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
|
| 118 |
+
"aggregation": "mean",
|
| 119 |
+
"higher_is_better": true
|
| 120 |
+
}
|
| 121 |
+
],
|
| 122 |
+
"output_type": "generate_until",
|
| 123 |
+
"generation_kwargs": {
|
| 124 |
+
"until": [
|
| 125 |
+
"[DONE]"
|
| 126 |
+
],
|
| 127 |
+
"do_sample": false
|
| 128 |
+
},
|
| 129 |
+
"repeats": 1,
|
| 130 |
+
"should_decontaminate": false,
|
| 131 |
+
"metadata": {
|
| 132 |
+
"version": 1.0,
|
| 133 |
+
"model": "student",
|
| 134 |
+
"base_url": "http://127.0.0.1:8422/v1/completions",
|
| 135 |
+
"tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0150",
|
| 136 |
+
"num_concurrent": 48,
|
| 137 |
+
"tokenized_requests": false,
|
| 138 |
+
"max_retries": 3,
|
| 139 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
|
| 140 |
+
}
|
| 141 |
+
}
|
| 142 |
+
},
|
| 143 |
+
"versions": {
|
| 144 |
+
"humaneval": 1.0,
|
| 145 |
+
"mbpp": 1.0
|
| 146 |
+
},
|
| 147 |
+
"n-shot": {
|
| 148 |
+
"humaneval": 0,
|
| 149 |
+
"mbpp": 3
|
| 150 |
+
},
|
| 151 |
+
"higher_is_better": {
|
| 152 |
+
"humaneval": {
|
| 153 |
+
"pass_at_k": true,
|
| 154 |
+
"pass@1": true
|
| 155 |
+
},
|
| 156 |
+
"mbpp": {
|
| 157 |
+
"pass_at_1": true
|
| 158 |
+
}
|
| 159 |
+
},
|
| 160 |
+
"n-samples": {
|
| 161 |
+
"humaneval": {
|
| 162 |
+
"original": 164,
|
| 163 |
+
"effective": 164
|
| 164 |
+
},
|
| 165 |
+
"mbpp": {
|
| 166 |
+
"original": 500,
|
| 167 |
+
"effective": 500
|
| 168 |
+
}
|
| 169 |
+
},
|
| 170 |
+
"config": {
|
| 171 |
+
"model": "local-completions",
|
| 172 |
+
"model_args": {
|
| 173 |
+
"model": "student",
|
| 174 |
+
"base_url": "http://127.0.0.1:8422/v1/completions",
|
| 175 |
+
"tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0150",
|
| 176 |
+
"num_concurrent": 48,
|
| 177 |
+
"tokenized_requests": false,
|
| 178 |
+
"max_retries": 3
|
| 179 |
+
},
|
| 180 |
+
"batch_size": 1,
|
| 181 |
+
"batch_sizes": [],
|
| 182 |
+
"device": "cuda:0",
|
| 183 |
+
"use_cache": null,
|
| 184 |
+
"limit": null,
|
| 185 |
+
"bootstrap_iters": 100000,
|
| 186 |
+
"gen_kwargs": {},
|
| 187 |
+
"random_seed": 0,
|
| 188 |
+
"numpy_seed": 1234,
|
| 189 |
+
"torch_seed": 1234,
|
| 190 |
+
"fewshot_seed": 1234
|
| 191 |
+
},
|
| 192 |
+
"git_hash": "4cd474a",
|
| 193 |
+
"date": 1784522214.7088265,
|
| 194 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 195 |
+
"transformers_version": "4.57.6",
|
| 196 |
+
"lm_eval_version": "0.4.12",
|
| 197 |
+
"upper_git_hash": null,
|
| 198 |
+
"tokenizer_pad_token": [
|
| 199 |
+
"<pad>",
|
| 200 |
+
"50280"
|
| 201 |
+
],
|
| 202 |
+
"tokenizer_eos_token": [
|
| 203 |
+
"|||IP_ADDRESS|||",
|
| 204 |
+
"50279"
|
| 205 |
+
],
|
| 206 |
+
"tokenizer_bos_token": [
|
| 207 |
+
"|||IP_ADDRESS|||",
|
| 208 |
+
"50279"
|
| 209 |
+
],
|
| 210 |
+
"eot_token_id": 50279,
|
| 211 |
+
"max_length": 2047,
|
| 212 |
+
"task_hashes": {
|
| 213 |
+
"humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
|
| 214 |
+
"mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
|
| 215 |
+
},
|
| 216 |
+
"model_source": "local-completions",
|
| 217 |
+
"model_name": "student",
|
| 218 |
+
"model_name_sanitized": "student",
|
| 219 |
+
"system_instruction": null,
|
| 220 |
+
"system_instruction_sha": null,
|
| 221 |
+
"fewshot_as_multiturn": null,
|
| 222 |
+
"chat_template": null,
|
| 223 |
+
"chat_template_sha": null,
|
| 224 |
+
"total_evaluation_time_seconds": "208.97489892301382"
|
| 225 |
+
}
|
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_gsm8k_cot_zeroshot_2026-07-19T21-36-45.489848.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_humaneval_2026-07-19T21-40-17.349569.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_ifeval_2026-07-19T21-36-45.489848.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_mbpp_2026-07-19T21-40-17.349569.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_minerva_math500_2026-07-19T21-36-45.489848.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/results_2026-07-19T21-49-11.345128.json
ADDED
|
@@ -0,0 +1,337 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"gsm8k_cot_zeroshot": {
|
| 4 |
+
"name": "gsm8k_cot_zeroshot",
|
| 5 |
+
"alias": "gsm8k_cot_zeroshot",
|
| 6 |
+
"sample_len": 1319,
|
| 7 |
+
"exact_match,strict-match": 0.0,
|
| 8 |
+
"exact_match_stderr,strict-match": 0.0,
|
| 9 |
+
"exact_match,flexible-extract": 0.558756633813495,
|
| 10 |
+
"exact_match_stderr,flexible-extract": 0.01367705947859264
|
| 11 |
+
},
|
| 12 |
+
"minerva_math500": {
|
| 13 |
+
"name": "minerva_math500",
|
| 14 |
+
"alias": "minerva_math500",
|
| 15 |
+
"sample_len": 500,
|
| 16 |
+
"exact_match,none": 0.138,
|
| 17 |
+
"exact_match_stderr,none": 0.015439843831953423,
|
| 18 |
+
"math_verify,none": 0.208,
|
| 19 |
+
"math_verify_stderr,none": 0.01816954222122996
|
| 20 |
+
},
|
| 21 |
+
"ifeval": {
|
| 22 |
+
"name": "ifeval",
|
| 23 |
+
"alias": "ifeval",
|
| 24 |
+
"sample_len": 541,
|
| 25 |
+
"prompt_level_strict_acc,none": 0.5730129390018485,
|
| 26 |
+
"prompt_level_strict_acc_stderr,none": 0.021285933050061243,
|
| 27 |
+
"inst_level_strict_acc,none": 0.6870503597122302,
|
| 28 |
+
"inst_level_strict_acc_stderr,none": "N/A",
|
| 29 |
+
"prompt_level_loose_acc,none": 0.5933456561922366,
|
| 30 |
+
"prompt_level_loose_acc_stderr,none": 0.021138283177336344,
|
| 31 |
+
"inst_level_loose_acc,none": 0.7086330935251799,
|
| 32 |
+
"inst_level_loose_acc_stderr,none": "N/A"
|
| 33 |
+
}
|
| 34 |
+
},
|
| 35 |
+
"group_subtasks": {},
|
| 36 |
+
"configs": {
|
| 37 |
+
"gsm8k_cot_zeroshot": {
|
| 38 |
+
"task": "gsm8k_cot_zeroshot",
|
| 39 |
+
"dataset_path": "openai/gsm8k",
|
| 40 |
+
"dataset_name": "main",
|
| 41 |
+
"training_split": "train",
|
| 42 |
+
"test_split": "test",
|
| 43 |
+
"fewshot_split": "train",
|
| 44 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 45 |
+
"doc_to_target": "{{answer}}",
|
| 46 |
+
"unsafe_code": false,
|
| 47 |
+
"description": "",
|
| 48 |
+
"target_delimiter": " ",
|
| 49 |
+
"fewshot_delimiter": "\n\n",
|
| 50 |
+
"fewshot_config": {
|
| 51 |
+
"sampler": "default",
|
| 52 |
+
"split": "train",
|
| 53 |
+
"process_docs": null,
|
| 54 |
+
"fewshot_indices": null,
|
| 55 |
+
"samples": null,
|
| 56 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 57 |
+
"doc_to_choice": null,
|
| 58 |
+
"doc_to_target": "{{answer}}",
|
| 59 |
+
"gen_prefix": null,
|
| 60 |
+
"fewshot_delimiter": "\n\n",
|
| 61 |
+
"target_delimiter": " "
|
| 62 |
+
},
|
| 63 |
+
"num_fewshot": 0,
|
| 64 |
+
"metric_list": [
|
| 65 |
+
{
|
| 66 |
+
"metric": "exact_match",
|
| 67 |
+
"aggregation": "mean",
|
| 68 |
+
"higher_is_better": true,
|
| 69 |
+
"ignore_case": true,
|
| 70 |
+
"ignore_punctuation": false,
|
| 71 |
+
"regexes_to_ignore": [
|
| 72 |
+
",",
|
| 73 |
+
"\\$",
|
| 74 |
+
"(?s).*#### ",
|
| 75 |
+
"\\.$"
|
| 76 |
+
]
|
| 77 |
+
}
|
| 78 |
+
],
|
| 79 |
+
"output_type": "generate_until",
|
| 80 |
+
"generation_kwargs": {
|
| 81 |
+
"until": [
|
| 82 |
+
"Q:",
|
| 83 |
+
"</s>",
|
| 84 |
+
"<|im_end|>"
|
| 85 |
+
],
|
| 86 |
+
"do_sample": false,
|
| 87 |
+
"max_gen_toks": 1280
|
| 88 |
+
},
|
| 89 |
+
"repeats": 1,
|
| 90 |
+
"filter_list": [
|
| 91 |
+
{
|
| 92 |
+
"name": "strict-match",
|
| 93 |
+
"filter": [
|
| 94 |
+
{
|
| 95 |
+
"function": "regex",
|
| 96 |
+
"regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"function": "take_first"
|
| 100 |
+
}
|
| 101 |
+
]
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"name": "flexible-extract",
|
| 105 |
+
"filter": [
|
| 106 |
+
{
|
| 107 |
+
"function": "regex",
|
| 108 |
+
"group_select": -1,
|
| 109 |
+
"regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"function": "take_first"
|
| 113 |
+
}
|
| 114 |
+
]
|
| 115 |
+
}
|
| 116 |
+
],
|
| 117 |
+
"should_decontaminate": false,
|
| 118 |
+
"metadata": {
|
| 119 |
+
"version": 3.0,
|
| 120 |
+
"model": "student",
|
| 121 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 122 |
+
"num_concurrent": 48,
|
| 123 |
+
"tokenized_requests": false,
|
| 124 |
+
"max_retries": 3,
|
| 125 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
|
| 126 |
+
}
|
| 127 |
+
},
|
| 128 |
+
"ifeval": {
|
| 129 |
+
"task": "ifeval",
|
| 130 |
+
"dataset_path": "google/IFEval",
|
| 131 |
+
"test_split": "train",
|
| 132 |
+
"doc_to_text": "prompt",
|
| 133 |
+
"doc_to_target": 0,
|
| 134 |
+
"unsafe_code": false,
|
| 135 |
+
"process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
|
| 136 |
+
"description": "",
|
| 137 |
+
"target_delimiter": " ",
|
| 138 |
+
"fewshot_delimiter": "\n\n",
|
| 139 |
+
"fewshot_config": {
|
| 140 |
+
"sampler": "default",
|
| 141 |
+
"split": null,
|
| 142 |
+
"process_docs": null,
|
| 143 |
+
"fewshot_indices": null,
|
| 144 |
+
"samples": null,
|
| 145 |
+
"doc_to_text": "prompt",
|
| 146 |
+
"doc_to_choice": null,
|
| 147 |
+
"doc_to_target": 0,
|
| 148 |
+
"gen_prefix": null,
|
| 149 |
+
"fewshot_delimiter": "\n\n",
|
| 150 |
+
"target_delimiter": " "
|
| 151 |
+
},
|
| 152 |
+
"num_fewshot": 0,
|
| 153 |
+
"metric_list": [
|
| 154 |
+
{
|
| 155 |
+
"metric": "prompt_level_strict_acc",
|
| 156 |
+
"aggregation": "mean",
|
| 157 |
+
"higher_is_better": true
|
| 158 |
+
},
|
| 159 |
+
{
|
| 160 |
+
"metric": "inst_level_strict_acc",
|
| 161 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 162 |
+
"higher_is_better": true
|
| 163 |
+
},
|
| 164 |
+
{
|
| 165 |
+
"metric": "prompt_level_loose_acc",
|
| 166 |
+
"aggregation": "mean",
|
| 167 |
+
"higher_is_better": true
|
| 168 |
+
},
|
| 169 |
+
{
|
| 170 |
+
"metric": "inst_level_loose_acc",
|
| 171 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 172 |
+
"higher_is_better": true
|
| 173 |
+
}
|
| 174 |
+
],
|
| 175 |
+
"output_type": "generate_until",
|
| 176 |
+
"generation_kwargs": {
|
| 177 |
+
"until": [],
|
| 178 |
+
"do_sample": false,
|
| 179 |
+
"temperature": 0.0,
|
| 180 |
+
"max_gen_toks": 1280
|
| 181 |
+
},
|
| 182 |
+
"repeats": 1,
|
| 183 |
+
"should_decontaminate": false,
|
| 184 |
+
"metadata": {
|
| 185 |
+
"version": 4.0,
|
| 186 |
+
"model": "student",
|
| 187 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 188 |
+
"num_concurrent": 48,
|
| 189 |
+
"tokenized_requests": false,
|
| 190 |
+
"max_retries": 3,
|
| 191 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
|
| 192 |
+
}
|
| 193 |
+
},
|
| 194 |
+
"minerva_math500": {
|
| 195 |
+
"task": "minerva_math500",
|
| 196 |
+
"dataset_path": "HuggingFaceH4/MATH-500",
|
| 197 |
+
"dataset_name": "default",
|
| 198 |
+
"test_split": "test",
|
| 199 |
+
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
|
| 200 |
+
"doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
|
| 201 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 202 |
+
"unsafe_code": false,
|
| 203 |
+
"process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
|
| 204 |
+
"description": "",
|
| 205 |
+
"target_delimiter": " ",
|
| 206 |
+
"fewshot_delimiter": "\n\n",
|
| 207 |
+
"fewshot_config": {
|
| 208 |
+
"sampler": "first_n",
|
| 209 |
+
"split": null,
|
| 210 |
+
"process_docs": "<function process_docs at 0x750976e35b20>",
|
| 211 |
+
"fewshot_indices": null,
|
| 212 |
+
"samples": "<function list_fewshot_samples at 0x750976e379c0>",
|
| 213 |
+
"doc_to_text": "<function doc_to_text at 0x75097c5fd800>",
|
| 214 |
+
"doc_to_choice": null,
|
| 215 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 216 |
+
"gen_prefix": null,
|
| 217 |
+
"fewshot_delimiter": "\n\n",
|
| 218 |
+
"target_delimiter": " "
|
| 219 |
+
},
|
| 220 |
+
"num_fewshot": 4,
|
| 221 |
+
"metric_list": [
|
| 222 |
+
{
|
| 223 |
+
"metric": "exact_match",
|
| 224 |
+
"aggregation": "mean",
|
| 225 |
+
"higher_is_better": true
|
| 226 |
+
},
|
| 227 |
+
{
|
| 228 |
+
"metric": "math_verify",
|
| 229 |
+
"aggregation": "mean",
|
| 230 |
+
"higher_is_better": true
|
| 231 |
+
}
|
| 232 |
+
],
|
| 233 |
+
"output_type": "generate_until",
|
| 234 |
+
"generation_kwargs": {
|
| 235 |
+
"until": [
|
| 236 |
+
"Problem:"
|
| 237 |
+
],
|
| 238 |
+
"do_sample": false,
|
| 239 |
+
"temperature": 0.0,
|
| 240 |
+
"max_gen_toks": 1280
|
| 241 |
+
},
|
| 242 |
+
"repeats": 1,
|
| 243 |
+
"should_decontaminate": false,
|
| 244 |
+
"metadata": {
|
| 245 |
+
"version": 3.0,
|
| 246 |
+
"model": "student",
|
| 247 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 248 |
+
"num_concurrent": 48,
|
| 249 |
+
"tokenized_requests": false,
|
| 250 |
+
"max_retries": 3,
|
| 251 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
|
| 252 |
+
}
|
| 253 |
+
}
|
| 254 |
+
},
|
| 255 |
+
"versions": {
|
| 256 |
+
"gsm8k_cot_zeroshot": 3.0,
|
| 257 |
+
"ifeval": 4.0,
|
| 258 |
+
"minerva_math500": 3.0
|
| 259 |
+
},
|
| 260 |
+
"n-shot": {
|
| 261 |
+
"gsm8k_cot_zeroshot": 0,
|
| 262 |
+
"ifeval": 0,
|
| 263 |
+
"minerva_math500": 4
|
| 264 |
+
},
|
| 265 |
+
"higher_is_better": {
|
| 266 |
+
"gsm8k_cot_zeroshot": {
|
| 267 |
+
"exact_match": true
|
| 268 |
+
},
|
| 269 |
+
"ifeval": {
|
| 270 |
+
"prompt_level_strict_acc": true,
|
| 271 |
+
"inst_level_strict_acc": true,
|
| 272 |
+
"prompt_level_loose_acc": true,
|
| 273 |
+
"inst_level_loose_acc": true
|
| 274 |
+
},
|
| 275 |
+
"minerva_math500": {
|
| 276 |
+
"exact_match": true,
|
| 277 |
+
"math_verify": true
|
| 278 |
+
}
|
| 279 |
+
},
|
| 280 |
+
"n-samples": {
|
| 281 |
+
"gsm8k_cot_zeroshot": {
|
| 282 |
+
"original": 1319,
|
| 283 |
+
"effective": 1319
|
| 284 |
+
},
|
| 285 |
+
"minerva_math500": {
|
| 286 |
+
"original": 500,
|
| 287 |
+
"effective": 500
|
| 288 |
+
},
|
| 289 |
+
"ifeval": {
|
| 290 |
+
"original": 541,
|
| 291 |
+
"effective": 541
|
| 292 |
+
}
|
| 293 |
+
},
|
| 294 |
+
"config": {
|
| 295 |
+
"model": "local-chat-completions",
|
| 296 |
+
"model_args": {
|
| 297 |
+
"model": "student",
|
| 298 |
+
"base_url": "http://127.0.0.1:8422/v1/chat/completions",
|
| 299 |
+
"num_concurrent": 48,
|
| 300 |
+
"tokenized_requests": false,
|
| 301 |
+
"max_retries": 3
|
| 302 |
+
},
|
| 303 |
+
"batch_size": 1,
|
| 304 |
+
"batch_sizes": [],
|
| 305 |
+
"device": "cuda:0",
|
| 306 |
+
"use_cache": null,
|
| 307 |
+
"limit": null,
|
| 308 |
+
"bootstrap_iters": 100000,
|
| 309 |
+
"gen_kwargs": {
|
| 310 |
+
"max_gen_toks": 1280
|
| 311 |
+
},
|
| 312 |
+
"random_seed": 0,
|
| 313 |
+
"numpy_seed": 1234,
|
| 314 |
+
"torch_seed": 1234,
|
| 315 |
+
"fewshot_seed": 1234
|
| 316 |
+
},
|
| 317 |
+
"git_hash": "4cd474a",
|
| 318 |
+
"date": 1784522456.8447104,
|
| 319 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 320 |
+
"transformers_version": "4.57.6",
|
| 321 |
+
"lm_eval_version": "0.4.12",
|
| 322 |
+
"upper_git_hash": null,
|
| 323 |
+
"task_hashes": {
|
| 324 |
+
"gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
|
| 325 |
+
"minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
|
| 326 |
+
"ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
|
| 327 |
+
},
|
| 328 |
+
"model_source": "local-chat-completions",
|
| 329 |
+
"model_name": "student",
|
| 330 |
+
"model_name_sanitized": "student",
|
| 331 |
+
"system_instruction": null,
|
| 332 |
+
"system_instruction_sha": null,
|
| 333 |
+
"fewshot_as_multiturn": true,
|
| 334 |
+
"chat_template": "",
|
| 335 |
+
"chat_template_sha": null,
|
| 336 |
+
"total_evaluation_time_seconds": "500.82202401198447"
|
| 337 |
+
}
|
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/results_2026-07-19T21-52-17.977663.json
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"humaneval": {
|
| 4 |
+
"name": "humaneval",
|
| 5 |
+
"alias": "humaneval",
|
| 6 |
+
"sample_len": 164,
|
| 7 |
+
"pass@1,create_test": 0.3170731707317073,
|
| 8 |
+
"pass@1_stderr,create_test": 0.03644794381282879
|
| 9 |
+
},
|
| 10 |
+
"mbpp": {
|
| 11 |
+
"name": "mbpp",
|
| 12 |
+
"alias": "mbpp",
|
| 13 |
+
"sample_len": 500,
|
| 14 |
+
"pass_at_1,none": 0.234,
|
| 15 |
+
"pass_at_1_stderr,none": 0.01895274156489368
|
| 16 |
+
}
|
| 17 |
+
},
|
| 18 |
+
"group_subtasks": {},
|
| 19 |
+
"configs": {
|
| 20 |
+
"humaneval": {
|
| 21 |
+
"task": "humaneval",
|
| 22 |
+
"dataset_path": "openai/openai_humaneval",
|
| 23 |
+
"test_split": "test",
|
| 24 |
+
"doc_to_text": "{{prompt}}",
|
| 25 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 26 |
+
"unsafe_code": true,
|
| 27 |
+
"description": "",
|
| 28 |
+
"target_delimiter": " ",
|
| 29 |
+
"fewshot_delimiter": "\n\n",
|
| 30 |
+
"fewshot_config": {
|
| 31 |
+
"sampler": "default",
|
| 32 |
+
"split": null,
|
| 33 |
+
"process_docs": null,
|
| 34 |
+
"fewshot_indices": null,
|
| 35 |
+
"samples": null,
|
| 36 |
+
"doc_to_text": "{{prompt}}",
|
| 37 |
+
"doc_to_choice": null,
|
| 38 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 39 |
+
"gen_prefix": null,
|
| 40 |
+
"fewshot_delimiter": "\n\n",
|
| 41 |
+
"target_delimiter": " "
|
| 42 |
+
},
|
| 43 |
+
"num_fewshot": 0,
|
| 44 |
+
"metric_list": [
|
| 45 |
+
{
|
| 46 |
+
"metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
|
| 47 |
+
"aggregation": "mean",
|
| 48 |
+
"higher_is_better": true,
|
| 49 |
+
"k": [
|
| 50 |
+
1
|
| 51 |
+
]
|
| 52 |
+
}
|
| 53 |
+
],
|
| 54 |
+
"output_type": "generate_until",
|
| 55 |
+
"generation_kwargs": {
|
| 56 |
+
"until": [
|
| 57 |
+
"\nclass",
|
| 58 |
+
"\ndef",
|
| 59 |
+
"\n#",
|
| 60 |
+
"\nif",
|
| 61 |
+
"\nprint"
|
| 62 |
+
],
|
| 63 |
+
"max_gen_toks": 1024,
|
| 64 |
+
"do_sample": false
|
| 65 |
+
},
|
| 66 |
+
"repeats": 1,
|
| 67 |
+
"filter_list": [
|
| 68 |
+
{
|
| 69 |
+
"name": "create_test",
|
| 70 |
+
"filter": [
|
| 71 |
+
{
|
| 72 |
+
"function": "custom",
|
| 73 |
+
"filter_fn": "<function build_predictions at 0x7de0f336eca0>"
|
| 74 |
+
}
|
| 75 |
+
]
|
| 76 |
+
}
|
| 77 |
+
],
|
| 78 |
+
"should_decontaminate": false,
|
| 79 |
+
"metadata": {
|
| 80 |
+
"version": 1.0,
|
| 81 |
+
"model": "student",
|
| 82 |
+
"base_url": "http://127.0.0.1:8422/v1/completions",
|
| 83 |
+
"tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0500",
|
| 84 |
+
"num_concurrent": 48,
|
| 85 |
+
"tokenized_requests": false,
|
| 86 |
+
"max_retries": 3,
|
| 87 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
|
| 88 |
+
}
|
| 89 |
+
},
|
| 90 |
+
"mbpp": {
|
| 91 |
+
"task": "mbpp",
|
| 92 |
+
"dataset_path": "google-research-datasets/mbpp",
|
| 93 |
+
"dataset_name": "full",
|
| 94 |
+
"test_split": "test",
|
| 95 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 96 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 97 |
+
"unsafe_code": true,
|
| 98 |
+
"description": "",
|
| 99 |
+
"target_delimiter": "",
|
| 100 |
+
"fewshot_delimiter": "\n\n",
|
| 101 |
+
"fewshot_config": {
|
| 102 |
+
"sampler": "first_n",
|
| 103 |
+
"split": null,
|
| 104 |
+
"process_docs": null,
|
| 105 |
+
"fewshot_indices": null,
|
| 106 |
+
"samples": "<function list_fewshot_samples at 0x7de1bc5079c0>",
|
| 107 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 108 |
+
"doc_to_choice": null,
|
| 109 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 110 |
+
"gen_prefix": null,
|
| 111 |
+
"fewshot_delimiter": "\n\n",
|
| 112 |
+
"target_delimiter": ""
|
| 113 |
+
},
|
| 114 |
+
"num_fewshot": 3,
|
| 115 |
+
"metric_list": [
|
| 116 |
+
{
|
| 117 |
+
"metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
|
| 118 |
+
"aggregation": "mean",
|
| 119 |
+
"higher_is_better": true
|
| 120 |
+
}
|
| 121 |
+
],
|
| 122 |
+
"output_type": "generate_until",
|
| 123 |
+
"generation_kwargs": {
|
| 124 |
+
"until": [
|
| 125 |
+
"[DONE]"
|
| 126 |
+
],
|
| 127 |
+
"do_sample": false
|
| 128 |
+
},
|
| 129 |
+
"repeats": 1,
|
| 130 |
+
"should_decontaminate": false,
|
| 131 |
+
"metadata": {
|
| 132 |
+
"version": 1.0,
|
| 133 |
+
"model": "student",
|
| 134 |
+
"base_url": "http://127.0.0.1:8422/v1/completions",
|
| 135 |
+
"tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0500",
|
| 136 |
+
"num_concurrent": 48,
|
| 137 |
+
"tokenized_requests": false,
|
| 138 |
+
"max_retries": 3,
|
| 139 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
|
| 140 |
+
}
|
| 141 |
+
}
|
| 142 |
+
},
|
| 143 |
+
"versions": {
|
| 144 |
+
"humaneval": 1.0,
|
| 145 |
+
"mbpp": 1.0
|
| 146 |
+
},
|
| 147 |
+
"n-shot": {
|
| 148 |
+
"humaneval": 0,
|
| 149 |
+
"mbpp": 3
|
| 150 |
+
},
|
| 151 |
+
"higher_is_better": {
|
| 152 |
+
"humaneval": {
|
| 153 |
+
"pass_at_k": true,
|
| 154 |
+
"pass@1": true
|
| 155 |
+
},
|
| 156 |
+
"mbpp": {
|
| 157 |
+
"pass_at_1": true
|
| 158 |
+
}
|
| 159 |
+
},
|
| 160 |
+
"n-samples": {
|
| 161 |
+
"humaneval": {
|
| 162 |
+
"original": 164,
|
| 163 |
+
"effective": 164
|
| 164 |
+
},
|
| 165 |
+
"mbpp": {
|
| 166 |
+
"original": 500,
|
| 167 |
+
"effective": 500
|
| 168 |
+
}
|
| 169 |
+
},
|
| 170 |
+
"config": {
|
| 171 |
+
"model": "local-completions",
|
| 172 |
+
"model_args": {
|
| 173 |
+
"model": "student",
|
| 174 |
+
"base_url": "http://127.0.0.1:8422/v1/completions",
|
| 175 |
+
"tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0500",
|
| 176 |
+
"num_concurrent": 48,
|
| 177 |
+
"tokenized_requests": false,
|
| 178 |
+
"max_retries": 3
|
| 179 |
+
},
|
| 180 |
+
"batch_size": 1,
|
| 181 |
+
"batch_sizes": [],
|
| 182 |
+
"device": "cuda:0",
|
| 183 |
+
"use_cache": null,
|
| 184 |
+
"limit": null,
|
| 185 |
+
"bootstrap_iters": 100000,
|
| 186 |
+
"gen_kwargs": {},
|
| 187 |
+
"random_seed": 0,
|
| 188 |
+
"numpy_seed": 1234,
|
| 189 |
+
"torch_seed": 1234,
|
| 190 |
+
"fewshot_seed": 1234
|
| 191 |
+
},
|
| 192 |
+
"git_hash": "4cd474a",
|
| 193 |
+
"date": 1784522960.7513723,
|
| 194 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 195 |
+
"transformers_version": "4.57.6",
|
| 196 |
+
"lm_eval_version": "0.4.12",
|
| 197 |
+
"upper_git_hash": null,
|
| 198 |
+
"tokenizer_pad_token": [
|
| 199 |
+
"<pad>",
|
| 200 |
+
"50280"
|
| 201 |
+
],
|
| 202 |
+
"tokenizer_eos_token": [
|
| 203 |
+
"|||IP_ADDRESS|||",
|
| 204 |
+
"50279"
|
| 205 |
+
],
|
| 206 |
+
"tokenizer_bos_token": [
|
| 207 |
+
"|||IP_ADDRESS|||",
|
| 208 |
+
"50279"
|
| 209 |
+
],
|
| 210 |
+
"eot_token_id": 50279,
|
| 211 |
+
"max_length": 2047,
|
| 212 |
+
"task_hashes": {
|
| 213 |
+
"humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
|
| 214 |
+
"mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
|
| 215 |
+
},
|
| 216 |
+
"model_source": "local-completions",
|
| 217 |
+
"model_name": "student",
|
| 218 |
+
"model_name_sanitized": "student",
|
| 219 |
+
"system_instruction": null,
|
| 220 |
+
"system_instruction_sha": null,
|
| 221 |
+
"fewshot_as_multiturn": null,
|
| 222 |
+
"chat_template": null,
|
| 223 |
+
"chat_template_sha": null,
|
| 224 |
+
"total_evaluation_time_seconds": "183.55893040599767"
|
| 225 |
+
}
|
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_gsm8k_cot_zeroshot_2026-07-19T21-49-11.345128.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_humaneval_2026-07-19T21-52-17.977663.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_ifeval_2026-07-19T21-49-11.345128.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_mbpp_2026-07-19T21-52-17.977663.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_minerva_math500_2026-07-19T21-49-11.345128.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/healed/glean_keep50_s1224_step100/student/results_2026-07-18T10-26-00.525964.json
ADDED
|
@@ -0,0 +1,337 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"gsm8k_cot_zeroshot": {
|
| 4 |
+
"name": "gsm8k_cot_zeroshot",
|
| 5 |
+
"alias": "gsm8k_cot_zeroshot",
|
| 6 |
+
"sample_len": 1319,
|
| 7 |
+
"exact_match,strict-match": 0.0,
|
| 8 |
+
"exact_match_stderr,strict-match": 0.0,
|
| 9 |
+
"exact_match,flexible-extract": 0.535253980288097,
|
| 10 |
+
"exact_match_stderr,flexible-extract": 0.013738207990177317
|
| 11 |
+
},
|
| 12 |
+
"minerva_math500": {
|
| 13 |
+
"name": "minerva_math500",
|
| 14 |
+
"alias": "minerva_math500",
|
| 15 |
+
"sample_len": 500,
|
| 16 |
+
"exact_match,none": 0.124,
|
| 17 |
+
"exact_match_stderr,none": 0.014754096608517476,
|
| 18 |
+
"math_verify,none": 0.186,
|
| 19 |
+
"math_verify_stderr,none": 0.01741880678058399
|
| 20 |
+
},
|
| 21 |
+
"ifeval": {
|
| 22 |
+
"name": "ifeval",
|
| 23 |
+
"alias": "ifeval",
|
| 24 |
+
"sample_len": 541,
|
| 25 |
+
"prompt_level_strict_acc,none": 0.5415896487985212,
|
| 26 |
+
"prompt_level_strict_acc_stderr,none": 0.021442010560476468,
|
| 27 |
+
"inst_level_strict_acc,none": 0.6642685851318945,
|
| 28 |
+
"inst_level_strict_acc_stderr,none": "N/A",
|
| 29 |
+
"prompt_level_loose_acc,none": 0.5748613678373382,
|
| 30 |
+
"prompt_level_loose_acc_stderr,none": 0.021274039805355655,
|
| 31 |
+
"inst_level_loose_acc,none": 0.6954436450839329,
|
| 32 |
+
"inst_level_loose_acc_stderr,none": "N/A"
|
| 33 |
+
}
|
| 34 |
+
},
|
| 35 |
+
"group_subtasks": {},
|
| 36 |
+
"configs": {
|
| 37 |
+
"gsm8k_cot_zeroshot": {
|
| 38 |
+
"task": "gsm8k_cot_zeroshot",
|
| 39 |
+
"dataset_path": "openai/gsm8k",
|
| 40 |
+
"dataset_name": "main",
|
| 41 |
+
"training_split": "train",
|
| 42 |
+
"test_split": "test",
|
| 43 |
+
"fewshot_split": "train",
|
| 44 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 45 |
+
"doc_to_target": "{{answer}}",
|
| 46 |
+
"unsafe_code": false,
|
| 47 |
+
"description": "",
|
| 48 |
+
"target_delimiter": " ",
|
| 49 |
+
"fewshot_delimiter": "\n\n",
|
| 50 |
+
"fewshot_config": {
|
| 51 |
+
"sampler": "default",
|
| 52 |
+
"split": "train",
|
| 53 |
+
"process_docs": null,
|
| 54 |
+
"fewshot_indices": null,
|
| 55 |
+
"samples": null,
|
| 56 |
+
"doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
|
| 57 |
+
"doc_to_choice": null,
|
| 58 |
+
"doc_to_target": "{{answer}}",
|
| 59 |
+
"gen_prefix": null,
|
| 60 |
+
"fewshot_delimiter": "\n\n",
|
| 61 |
+
"target_delimiter": " "
|
| 62 |
+
},
|
| 63 |
+
"num_fewshot": 0,
|
| 64 |
+
"metric_list": [
|
| 65 |
+
{
|
| 66 |
+
"metric": "exact_match",
|
| 67 |
+
"aggregation": "mean",
|
| 68 |
+
"higher_is_better": true,
|
| 69 |
+
"ignore_case": true,
|
| 70 |
+
"ignore_punctuation": false,
|
| 71 |
+
"regexes_to_ignore": [
|
| 72 |
+
",",
|
| 73 |
+
"\\$",
|
| 74 |
+
"(?s).*#### ",
|
| 75 |
+
"\\.$"
|
| 76 |
+
]
|
| 77 |
+
}
|
| 78 |
+
],
|
| 79 |
+
"output_type": "generate_until",
|
| 80 |
+
"generation_kwargs": {
|
| 81 |
+
"until": [
|
| 82 |
+
"Q:",
|
| 83 |
+
"</s>",
|
| 84 |
+
"<|im_end|>"
|
| 85 |
+
],
|
| 86 |
+
"do_sample": false,
|
| 87 |
+
"max_gen_toks": 1280
|
| 88 |
+
},
|
| 89 |
+
"repeats": 1,
|
| 90 |
+
"filter_list": [
|
| 91 |
+
{
|
| 92 |
+
"name": "strict-match",
|
| 93 |
+
"filter": [
|
| 94 |
+
{
|
| 95 |
+
"function": "regex",
|
| 96 |
+
"regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"function": "take_first"
|
| 100 |
+
}
|
| 101 |
+
]
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"name": "flexible-extract",
|
| 105 |
+
"filter": [
|
| 106 |
+
{
|
| 107 |
+
"function": "regex",
|
| 108 |
+
"group_select": -1,
|
| 109 |
+
"regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"function": "take_first"
|
| 113 |
+
}
|
| 114 |
+
]
|
| 115 |
+
}
|
| 116 |
+
],
|
| 117 |
+
"should_decontaminate": false,
|
| 118 |
+
"metadata": {
|
| 119 |
+
"version": 3.0,
|
| 120 |
+
"model": "student",
|
| 121 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 122 |
+
"num_concurrent": 48,
|
| 123 |
+
"tokenized_requests": false,
|
| 124 |
+
"max_retries": 3,
|
| 125 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
|
| 126 |
+
}
|
| 127 |
+
},
|
| 128 |
+
"ifeval": {
|
| 129 |
+
"task": "ifeval",
|
| 130 |
+
"dataset_path": "google/IFEval",
|
| 131 |
+
"test_split": "train",
|
| 132 |
+
"doc_to_text": "prompt",
|
| 133 |
+
"doc_to_target": 0,
|
| 134 |
+
"unsafe_code": false,
|
| 135 |
+
"process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
|
| 136 |
+
"description": "",
|
| 137 |
+
"target_delimiter": " ",
|
| 138 |
+
"fewshot_delimiter": "\n\n",
|
| 139 |
+
"fewshot_config": {
|
| 140 |
+
"sampler": "default",
|
| 141 |
+
"split": null,
|
| 142 |
+
"process_docs": null,
|
| 143 |
+
"fewshot_indices": null,
|
| 144 |
+
"samples": null,
|
| 145 |
+
"doc_to_text": "prompt",
|
| 146 |
+
"doc_to_choice": null,
|
| 147 |
+
"doc_to_target": 0,
|
| 148 |
+
"gen_prefix": null,
|
| 149 |
+
"fewshot_delimiter": "\n\n",
|
| 150 |
+
"target_delimiter": " "
|
| 151 |
+
},
|
| 152 |
+
"num_fewshot": 0,
|
| 153 |
+
"metric_list": [
|
| 154 |
+
{
|
| 155 |
+
"metric": "prompt_level_strict_acc",
|
| 156 |
+
"aggregation": "mean",
|
| 157 |
+
"higher_is_better": true
|
| 158 |
+
},
|
| 159 |
+
{
|
| 160 |
+
"metric": "inst_level_strict_acc",
|
| 161 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 162 |
+
"higher_is_better": true
|
| 163 |
+
},
|
| 164 |
+
{
|
| 165 |
+
"metric": "prompt_level_loose_acc",
|
| 166 |
+
"aggregation": "mean",
|
| 167 |
+
"higher_is_better": true
|
| 168 |
+
},
|
| 169 |
+
{
|
| 170 |
+
"metric": "inst_level_loose_acc",
|
| 171 |
+
"aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
|
| 172 |
+
"higher_is_better": true
|
| 173 |
+
}
|
| 174 |
+
],
|
| 175 |
+
"output_type": "generate_until",
|
| 176 |
+
"generation_kwargs": {
|
| 177 |
+
"until": [],
|
| 178 |
+
"do_sample": false,
|
| 179 |
+
"temperature": 0.0,
|
| 180 |
+
"max_gen_toks": 1280
|
| 181 |
+
},
|
| 182 |
+
"repeats": 1,
|
| 183 |
+
"should_decontaminate": false,
|
| 184 |
+
"metadata": {
|
| 185 |
+
"version": 4.0,
|
| 186 |
+
"model": "student",
|
| 187 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 188 |
+
"num_concurrent": 48,
|
| 189 |
+
"tokenized_requests": false,
|
| 190 |
+
"max_retries": 3,
|
| 191 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
|
| 192 |
+
}
|
| 193 |
+
},
|
| 194 |
+
"minerva_math500": {
|
| 195 |
+
"task": "minerva_math500",
|
| 196 |
+
"dataset_path": "HuggingFaceH4/MATH-500",
|
| 197 |
+
"dataset_name": "default",
|
| 198 |
+
"test_split": "test",
|
| 199 |
+
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
|
| 200 |
+
"doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
|
| 201 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 202 |
+
"unsafe_code": false,
|
| 203 |
+
"process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
|
| 204 |
+
"description": "",
|
| 205 |
+
"target_delimiter": " ",
|
| 206 |
+
"fewshot_delimiter": "\n\n",
|
| 207 |
+
"fewshot_config": {
|
| 208 |
+
"sampler": "first_n",
|
| 209 |
+
"split": null,
|
| 210 |
+
"process_docs": "<function process_docs at 0x7f32b8039b20>",
|
| 211 |
+
"fewshot_indices": null,
|
| 212 |
+
"samples": "<function list_fewshot_samples at 0x7f32b803b9c0>",
|
| 213 |
+
"doc_to_text": "<function doc_to_text at 0x7f32b9601800>",
|
| 214 |
+
"doc_to_choice": null,
|
| 215 |
+
"doc_to_target": "{{answer if few_shot is undefined else solution}}",
|
| 216 |
+
"gen_prefix": null,
|
| 217 |
+
"fewshot_delimiter": "\n\n",
|
| 218 |
+
"target_delimiter": " "
|
| 219 |
+
},
|
| 220 |
+
"num_fewshot": 4,
|
| 221 |
+
"metric_list": [
|
| 222 |
+
{
|
| 223 |
+
"metric": "exact_match",
|
| 224 |
+
"aggregation": "mean",
|
| 225 |
+
"higher_is_better": true
|
| 226 |
+
},
|
| 227 |
+
{
|
| 228 |
+
"metric": "math_verify",
|
| 229 |
+
"aggregation": "mean",
|
| 230 |
+
"higher_is_better": true
|
| 231 |
+
}
|
| 232 |
+
],
|
| 233 |
+
"output_type": "generate_until",
|
| 234 |
+
"generation_kwargs": {
|
| 235 |
+
"until": [
|
| 236 |
+
"Problem:"
|
| 237 |
+
],
|
| 238 |
+
"do_sample": false,
|
| 239 |
+
"temperature": 0.0,
|
| 240 |
+
"max_gen_toks": 1280
|
| 241 |
+
},
|
| 242 |
+
"repeats": 1,
|
| 243 |
+
"should_decontaminate": false,
|
| 244 |
+
"metadata": {
|
| 245 |
+
"version": 3.0,
|
| 246 |
+
"model": "student",
|
| 247 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 248 |
+
"num_concurrent": 48,
|
| 249 |
+
"tokenized_requests": false,
|
| 250 |
+
"max_retries": 3,
|
| 251 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
|
| 252 |
+
}
|
| 253 |
+
}
|
| 254 |
+
},
|
| 255 |
+
"versions": {
|
| 256 |
+
"gsm8k_cot_zeroshot": 3.0,
|
| 257 |
+
"ifeval": 4.0,
|
| 258 |
+
"minerva_math500": 3.0
|
| 259 |
+
},
|
| 260 |
+
"n-shot": {
|
| 261 |
+
"gsm8k_cot_zeroshot": 0,
|
| 262 |
+
"ifeval": 0,
|
| 263 |
+
"minerva_math500": 4
|
| 264 |
+
},
|
| 265 |
+
"higher_is_better": {
|
| 266 |
+
"gsm8k_cot_zeroshot": {
|
| 267 |
+
"exact_match": true
|
| 268 |
+
},
|
| 269 |
+
"ifeval": {
|
| 270 |
+
"prompt_level_strict_acc": true,
|
| 271 |
+
"inst_level_strict_acc": true,
|
| 272 |
+
"prompt_level_loose_acc": true,
|
| 273 |
+
"inst_level_loose_acc": true
|
| 274 |
+
},
|
| 275 |
+
"minerva_math500": {
|
| 276 |
+
"exact_match": true,
|
| 277 |
+
"math_verify": true
|
| 278 |
+
}
|
| 279 |
+
},
|
| 280 |
+
"n-samples": {
|
| 281 |
+
"gsm8k_cot_zeroshot": {
|
| 282 |
+
"original": 1319,
|
| 283 |
+
"effective": 1319
|
| 284 |
+
},
|
| 285 |
+
"minerva_math500": {
|
| 286 |
+
"original": 500,
|
| 287 |
+
"effective": 500
|
| 288 |
+
},
|
| 289 |
+
"ifeval": {
|
| 290 |
+
"original": 541,
|
| 291 |
+
"effective": 541
|
| 292 |
+
}
|
| 293 |
+
},
|
| 294 |
+
"config": {
|
| 295 |
+
"model": "local-chat-completions",
|
| 296 |
+
"model_args": {
|
| 297 |
+
"model": "student",
|
| 298 |
+
"base_url": "http://127.0.0.1:8420/v1/chat/completions",
|
| 299 |
+
"num_concurrent": 48,
|
| 300 |
+
"tokenized_requests": false,
|
| 301 |
+
"max_retries": 3
|
| 302 |
+
},
|
| 303 |
+
"batch_size": 1,
|
| 304 |
+
"batch_sizes": [],
|
| 305 |
+
"device": "cuda:0",
|
| 306 |
+
"use_cache": null,
|
| 307 |
+
"limit": null,
|
| 308 |
+
"bootstrap_iters": 100000,
|
| 309 |
+
"gen_kwargs": {
|
| 310 |
+
"max_gen_toks": 1280
|
| 311 |
+
},
|
| 312 |
+
"random_seed": 0,
|
| 313 |
+
"numpy_seed": 1234,
|
| 314 |
+
"torch_seed": 1234,
|
| 315 |
+
"fewshot_seed": 1234
|
| 316 |
+
},
|
| 317 |
+
"git_hash": "4cd474a",
|
| 318 |
+
"date": 1784394960.8878376,
|
| 319 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 320 |
+
"transformers_version": "4.57.6",
|
| 321 |
+
"lm_eval_version": "0.4.12",
|
| 322 |
+
"upper_git_hash": null,
|
| 323 |
+
"task_hashes": {
|
| 324 |
+
"gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
|
| 325 |
+
"minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
|
| 326 |
+
"ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
|
| 327 |
+
},
|
| 328 |
+
"model_source": "local-chat-completions",
|
| 329 |
+
"model_name": "student",
|
| 330 |
+
"model_name_sanitized": "student",
|
| 331 |
+
"system_instruction": null,
|
| 332 |
+
"system_instruction_sha": null,
|
| 333 |
+
"fewshot_as_multiturn": true,
|
| 334 |
+
"chat_template": "",
|
| 335 |
+
"chat_template_sha": null,
|
| 336 |
+
"total_evaluation_time_seconds": "605.9618037229993"
|
| 337 |
+
}
|
evals/general_suite/healed/glean_keep50_s1224_step100/student/results_2026-07-18T10-29-19.004514.json
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"humaneval": {
|
| 4 |
+
"name": "humaneval",
|
| 5 |
+
"alias": "humaneval",
|
| 6 |
+
"sample_len": 164,
|
| 7 |
+
"pass@1,create_test": 0.29878048780487804,
|
| 8 |
+
"pass@1_stderr,create_test": 0.035851663369096606
|
| 9 |
+
},
|
| 10 |
+
"mbpp": {
|
| 11 |
+
"name": "mbpp",
|
| 12 |
+
"alias": "mbpp",
|
| 13 |
+
"sample_len": 500,
|
| 14 |
+
"pass_at_1,none": 0.218,
|
| 15 |
+
"pass_at_1_stderr,none": 0.018483378223178866
|
| 16 |
+
}
|
| 17 |
+
},
|
| 18 |
+
"group_subtasks": {},
|
| 19 |
+
"configs": {
|
| 20 |
+
"humaneval": {
|
| 21 |
+
"task": "humaneval",
|
| 22 |
+
"dataset_path": "openai/openai_humaneval",
|
| 23 |
+
"test_split": "test",
|
| 24 |
+
"doc_to_text": "{{prompt}}",
|
| 25 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 26 |
+
"unsafe_code": true,
|
| 27 |
+
"description": "",
|
| 28 |
+
"target_delimiter": " ",
|
| 29 |
+
"fewshot_delimiter": "\n\n",
|
| 30 |
+
"fewshot_config": {
|
| 31 |
+
"sampler": "default",
|
| 32 |
+
"split": null,
|
| 33 |
+
"process_docs": null,
|
| 34 |
+
"fewshot_indices": null,
|
| 35 |
+
"samples": null,
|
| 36 |
+
"doc_to_text": "{{prompt}}",
|
| 37 |
+
"doc_to_choice": null,
|
| 38 |
+
"doc_to_target": "{{test}}\ncheck({{entry_point}})",
|
| 39 |
+
"gen_prefix": null,
|
| 40 |
+
"fewshot_delimiter": "\n\n",
|
| 41 |
+
"target_delimiter": " "
|
| 42 |
+
},
|
| 43 |
+
"num_fewshot": 0,
|
| 44 |
+
"metric_list": [
|
| 45 |
+
{
|
| 46 |
+
"metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
|
| 47 |
+
"aggregation": "mean",
|
| 48 |
+
"higher_is_better": true,
|
| 49 |
+
"k": [
|
| 50 |
+
1
|
| 51 |
+
]
|
| 52 |
+
}
|
| 53 |
+
],
|
| 54 |
+
"output_type": "generate_until",
|
| 55 |
+
"generation_kwargs": {
|
| 56 |
+
"until": [
|
| 57 |
+
"\nclass",
|
| 58 |
+
"\ndef",
|
| 59 |
+
"\n#",
|
| 60 |
+
"\nif",
|
| 61 |
+
"\nprint"
|
| 62 |
+
],
|
| 63 |
+
"max_gen_toks": 1024,
|
| 64 |
+
"do_sample": false
|
| 65 |
+
},
|
| 66 |
+
"repeats": 1,
|
| 67 |
+
"filter_list": [
|
| 68 |
+
{
|
| 69 |
+
"name": "create_test",
|
| 70 |
+
"filter": [
|
| 71 |
+
{
|
| 72 |
+
"function": "custom",
|
| 73 |
+
"filter_fn": "<function build_predictions at 0x75dd03f1eca0>"
|
| 74 |
+
}
|
| 75 |
+
]
|
| 76 |
+
}
|
| 77 |
+
],
|
| 78 |
+
"should_decontaminate": false,
|
| 79 |
+
"metadata": {
|
| 80 |
+
"version": 1.0,
|
| 81 |
+
"model": "student",
|
| 82 |
+
"base_url": "http://127.0.0.1:8420/v1/completions",
|
| 83 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0100",
|
| 84 |
+
"num_concurrent": 48,
|
| 85 |
+
"tokenized_requests": false,
|
| 86 |
+
"max_retries": 3,
|
| 87 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
|
| 88 |
+
}
|
| 89 |
+
},
|
| 90 |
+
"mbpp": {
|
| 91 |
+
"task": "mbpp",
|
| 92 |
+
"dataset_path": "google-research-datasets/mbpp",
|
| 93 |
+
"dataset_name": "full",
|
| 94 |
+
"test_split": "test",
|
| 95 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 96 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 97 |
+
"unsafe_code": true,
|
| 98 |
+
"description": "",
|
| 99 |
+
"target_delimiter": "",
|
| 100 |
+
"fewshot_delimiter": "\n\n",
|
| 101 |
+
"fewshot_config": {
|
| 102 |
+
"sampler": "first_n",
|
| 103 |
+
"split": null,
|
| 104 |
+
"process_docs": null,
|
| 105 |
+
"fewshot_indices": null,
|
| 106 |
+
"samples": "<function list_fewshot_samples at 0x75ddcceff9c0>",
|
| 107 |
+
"doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
|
| 108 |
+
"doc_to_choice": null,
|
| 109 |
+
"doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
|
| 110 |
+
"gen_prefix": null,
|
| 111 |
+
"fewshot_delimiter": "\n\n",
|
| 112 |
+
"target_delimiter": ""
|
| 113 |
+
},
|
| 114 |
+
"num_fewshot": 3,
|
| 115 |
+
"metric_list": [
|
| 116 |
+
{
|
| 117 |
+
"metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
|
| 118 |
+
"aggregation": "mean",
|
| 119 |
+
"higher_is_better": true
|
| 120 |
+
}
|
| 121 |
+
],
|
| 122 |
+
"output_type": "generate_until",
|
| 123 |
+
"generation_kwargs": {
|
| 124 |
+
"until": [
|
| 125 |
+
"[DONE]"
|
| 126 |
+
],
|
| 127 |
+
"do_sample": false
|
| 128 |
+
},
|
| 129 |
+
"repeats": 1,
|
| 130 |
+
"should_decontaminate": false,
|
| 131 |
+
"metadata": {
|
| 132 |
+
"version": 1.0,
|
| 133 |
+
"model": "student",
|
| 134 |
+
"base_url": "http://127.0.0.1:8420/v1/completions",
|
| 135 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0100",
|
| 136 |
+
"num_concurrent": 48,
|
| 137 |
+
"tokenized_requests": false,
|
| 138 |
+
"max_retries": 3,
|
| 139 |
+
"config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
|
| 140 |
+
}
|
| 141 |
+
}
|
| 142 |
+
},
|
| 143 |
+
"versions": {
|
| 144 |
+
"humaneval": 1.0,
|
| 145 |
+
"mbpp": 1.0
|
| 146 |
+
},
|
| 147 |
+
"n-shot": {
|
| 148 |
+
"humaneval": 0,
|
| 149 |
+
"mbpp": 3
|
| 150 |
+
},
|
| 151 |
+
"higher_is_better": {
|
| 152 |
+
"humaneval": {
|
| 153 |
+
"pass_at_k": true,
|
| 154 |
+
"pass@1": true
|
| 155 |
+
},
|
| 156 |
+
"mbpp": {
|
| 157 |
+
"pass_at_1": true
|
| 158 |
+
}
|
| 159 |
+
},
|
| 160 |
+
"n-samples": {
|
| 161 |
+
"humaneval": {
|
| 162 |
+
"original": 164,
|
| 163 |
+
"effective": 164
|
| 164 |
+
},
|
| 165 |
+
"mbpp": {
|
| 166 |
+
"original": 500,
|
| 167 |
+
"effective": 500
|
| 168 |
+
}
|
| 169 |
+
},
|
| 170 |
+
"config": {
|
| 171 |
+
"model": "local-completions",
|
| 172 |
+
"model_args": {
|
| 173 |
+
"model": "student",
|
| 174 |
+
"base_url": "http://127.0.0.1:8420/v1/completions",
|
| 175 |
+
"tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0100",
|
| 176 |
+
"num_concurrent": 48,
|
| 177 |
+
"tokenized_requests": false,
|
| 178 |
+
"max_retries": 3
|
| 179 |
+
},
|
| 180 |
+
"batch_size": 1,
|
| 181 |
+
"batch_sizes": [],
|
| 182 |
+
"device": "cuda:0",
|
| 183 |
+
"use_cache": null,
|
| 184 |
+
"limit": null,
|
| 185 |
+
"bootstrap_iters": 100000,
|
| 186 |
+
"gen_kwargs": {},
|
| 187 |
+
"random_seed": 0,
|
| 188 |
+
"numpy_seed": 1234,
|
| 189 |
+
"torch_seed": 1234,
|
| 190 |
+
"fewshot_seed": 1234
|
| 191 |
+
},
|
| 192 |
+
"git_hash": "4cd474a",
|
| 193 |
+
"date": 1784395569.954176,
|
| 194 |
+
"pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
|
| 195 |
+
"transformers_version": "4.57.6",
|
| 196 |
+
"lm_eval_version": "0.4.12",
|
| 197 |
+
"upper_git_hash": null,
|
| 198 |
+
"tokenizer_pad_token": [
|
| 199 |
+
"<pad>",
|
| 200 |
+
"50280"
|
| 201 |
+
],
|
| 202 |
+
"tokenizer_eos_token": [
|
| 203 |
+
"|||IP_ADDRESS|||",
|
| 204 |
+
"50279"
|
| 205 |
+
],
|
| 206 |
+
"tokenizer_bos_token": [
|
| 207 |
+
"|||IP_ADDRESS|||",
|
| 208 |
+
"50279"
|
| 209 |
+
],
|
| 210 |
+
"eot_token_id": 50279,
|
| 211 |
+
"max_length": 2047,
|
| 212 |
+
"task_hashes": {
|
| 213 |
+
"humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
|
| 214 |
+
"mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
|
| 215 |
+
},
|
| 216 |
+
"model_source": "local-completions",
|
| 217 |
+
"model_name": "student",
|
| 218 |
+
"model_name_sanitized": "student",
|
| 219 |
+
"system_instruction": null,
|
| 220 |
+
"system_instruction_sha": null,
|
| 221 |
+
"fewshot_as_multiturn": null,
|
| 222 |
+
"chat_template": null,
|
| 223 |
+
"chat_template_sha": null,
|
| 224 |
+
"total_evaluation_time_seconds": "195.39144668500012"
|
| 225 |
+
}
|