hbfreed commited on
Commit
9057161
·
verified ·
1 Parent(s): a050468

Add files using upload-large-folder tool

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. evals/general_suite/healed/glean_keep25_s1224/lm_eval.log +0 -0
  2. evals/general_suite/healed/glean_keep25_s1224/lm_eval_code.log +27 -0
  3. evals/general_suite/healed/glean_keep25_s1224/server.log +0 -0
  4. evals/general_suite/healed/glean_keep25_s1224/student/results_2026-07-17T14-56-42.779609.json +337 -0
  5. evals/general_suite/healed/glean_keep25_s1224/student/results_2026-07-17T15-01-00.225834.json +225 -0
  6. evals/general_suite/healed/glean_keep25_s1224/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-56-42.779609.jsonl +0 -0
  7. evals/general_suite/healed/glean_keep25_s1224/student/samples_humaneval_2026-07-17T15-01-00.225834.jsonl +0 -0
  8. evals/general_suite/healed/glean_keep25_s1224/student/samples_ifeval_2026-07-17T14-56-42.779609.jsonl +0 -0
  9. evals/general_suite/healed/glean_keep25_s1224/student/samples_mbpp_2026-07-17T15-01-00.225834.jsonl +0 -0
  10. evals/general_suite/healed/glean_keep25_s1224/student/samples_minerva_math500_2026-07-17T14-56-42.779609.jsonl +0 -0
  11. evals/general_suite/healed/glean_keep25_s1225/student/results_2026-07-17T14-48-07.537542.json +337 -0
  12. evals/general_suite/healed/glean_keep25_s1225/student/results_2026-07-17T14-54-29.755565.json +225 -0
  13. evals/general_suite/healed/glean_keep25_s1225/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-48-07.537542.jsonl +0 -0
  14. evals/general_suite/healed/glean_keep25_s1225/student/samples_humaneval_2026-07-17T14-54-29.755565.jsonl +0 -0
  15. evals/general_suite/healed/glean_keep25_s1225/student/samples_ifeval_2026-07-17T14-48-07.537542.jsonl +0 -0
  16. evals/general_suite/healed/glean_keep25_s1225/student/samples_mbpp_2026-07-17T14-54-29.755565.jsonl +0 -0
  17. evals/general_suite/healed/glean_keep25_s1225/student/samples_minerva_math500_2026-07-17T14-48-07.537542.jsonl +0 -0
  18. evals/general_suite/healed/glean_keep25_s1226/lm_eval.log +0 -0
  19. evals/general_suite/healed/glean_keep25_s1226/lm_eval_code.log +27 -0
  20. evals/general_suite/healed/glean_keep25_s1226/server.log +0 -0
  21. evals/general_suite/healed/glean_keep25_s1226/student/results_2026-07-17T14-48-34.159099.json +337 -0
  22. evals/general_suite/healed/glean_keep25_s1226/student/results_2026-07-17T14-54-44.918246.json +225 -0
  23. evals/general_suite/healed/glean_keep25_s1226/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-48-34.159099.jsonl +0 -0
  24. evals/general_suite/healed/glean_keep25_s1226/student/samples_humaneval_2026-07-17T14-54-44.918246.jsonl +0 -0
  25. evals/general_suite/healed/glean_keep25_s1226/student/samples_ifeval_2026-07-17T14-48-34.159099.jsonl +0 -0
  26. evals/general_suite/healed/glean_keep25_s1226/student/samples_mbpp_2026-07-17T14-54-44.918246.jsonl +0 -0
  27. evals/general_suite/healed/glean_keep25_s1226/student/samples_minerva_math500_2026-07-17T14-48-34.159099.jsonl +0 -0
  28. evals/general_suite/healed/glean_keep50_s1224/student/results_2026-07-18T10-11-54.989656.json +337 -0
  29. evals/general_suite/healed/glean_keep50_s1224/student/results_2026-07-18T10-15-21.380553.json +225 -0
  30. evals/general_suite/healed/glean_keep50_s1224/student/samples_gsm8k_cot_zeroshot_2026-07-18T10-11-54.989656.jsonl +0 -0
  31. evals/general_suite/healed/glean_keep50_s1224/student/samples_humaneval_2026-07-18T10-15-21.380553.jsonl +0 -0
  32. evals/general_suite/healed/glean_keep50_s1224/student/samples_ifeval_2026-07-18T10-11-54.989656.jsonl +0 -0
  33. evals/general_suite/healed/glean_keep50_s1224/student/samples_mbpp_2026-07-18T10-15-21.380553.jsonl +0 -0
  34. evals/general_suite/healed/glean_keep50_s1224/student/samples_minerva_math500_2026-07-18T10-11-54.989656.jsonl +0 -0
  35. evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/results_2026-07-19T21-36-45.489848.json +337 -0
  36. evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/results_2026-07-19T21-40-17.349569.json +225 -0
  37. evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_gsm8k_cot_zeroshot_2026-07-19T21-36-45.489848.jsonl +0 -0
  38. evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_humaneval_2026-07-19T21-40-17.349569.jsonl +0 -0
  39. evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_ifeval_2026-07-19T21-36-45.489848.jsonl +0 -0
  40. evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_mbpp_2026-07-19T21-40-17.349569.jsonl +0 -0
  41. evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_minerva_math500_2026-07-19T21-36-45.489848.jsonl +0 -0
  42. evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/results_2026-07-19T21-49-11.345128.json +337 -0
  43. evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/results_2026-07-19T21-52-17.977663.json +225 -0
  44. evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_gsm8k_cot_zeroshot_2026-07-19T21-49-11.345128.jsonl +0 -0
  45. evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_humaneval_2026-07-19T21-52-17.977663.jsonl +0 -0
  46. evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_ifeval_2026-07-19T21-49-11.345128.jsonl +0 -0
  47. evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_mbpp_2026-07-19T21-52-17.977663.jsonl +0 -0
  48. evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_minerva_math500_2026-07-19T21-49-11.345128.jsonl +0 -0
  49. evals/general_suite/healed/glean_keep50_s1224_step100/student/results_2026-07-18T10-26-00.525964.json +337 -0
  50. evals/general_suite/healed/glean_keep50_s1224_step100/student/results_2026-07-18T10-29-19.004514.json +225 -0
evals/general_suite/healed/glean_keep25_s1224/lm_eval.log ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1224/lm_eval_code.log ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/164 [00:00<?, ?it/s]
1
  89%|████████▉ | 146/164 [00:00<00:00, 1454.83it/s]
 
 
2
  0%| | 0/500 [00:00<?, ?it/s]
3
  3%|▎ | 17/500 [00:00<00:02, 162.50it/s]
4
  7%|▋ | 34/500 [00:00<00:02, 162.43it/s]
5
  10%|█ | 51/500 [00:00<00:02, 163.22it/s]
6
  14%|█▎ | 68/500 [00:00<00:02, 163.57it/s]
7
  17%|█▋ | 85/500 [00:00<00:02, 164.04it/s]
8
  20%|██ | 102/500 [00:00<00:02, 164.63it/s]
9
  24%|██▍ | 119/500 [00:00<00:02, 165.03it/s]
10
  27%|██▋ | 136/500 [00:00<00:02, 164.92it/s]
11
  31%|███ | 153/500 [00:00<00:02, 164.85it/s]
12
  34%|███▍ | 170/500 [00:01<00:01, 165.15it/s]
13
  37%|███▋ | 187/500 [00:01<00:01, 165.38it/s]
14
  41%|████ | 204/500 [00:01<00:01, 165.09it/s]
15
  44%|████▍ | 221/500 [00:01<00:01, 165.27it/s]
16
  48%|████▊ | 238/500 [00:01<00:01, 165.35it/s]
17
  51%|█████ | 255/500 [00:01<00:01, 165.52it/s]
18
  54%|█████▍ | 272/500 [00:01<00:01, 165.75it/s]
19
  58%|█████▊ | 289/500 [00:01<00:01, 165.94it/s]
20
  61%|██████ | 306/500 [00:01<00:01, 165.69it/s]
21
  65%|██████▍ | 323/500 [00:01<00:01, 165.94it/s]
22
  68%|██████▊ | 340/500 [00:02<00:00, 166.00it/s]
23
  71%|███████▏ | 357/500 [00:02<00:00, 166.04it/s]
24
  75%|███████▍ | 374/500 [00:02<00:00, 165.69it/s]
25
  78%|███████▊ | 391/500 [00:02<00:00, 165.72it/s]
26
  82%|████████▏ | 408/500 [00:02<00:00, 165.76it/s]
27
  85%|████████▌ | 425/500 [00:02<00:00, 165.87it/s]
28
  88%|████████▊ | 442/500 [00:02<00:00, 166.13it/s]
29
  92%|█████████▏| 459/500 [00:02<00:00, 166.11it/s]
30
  95%|█████████▌| 476/500 [00:02<00:00, 165.83it/s]
31
  99%|█████████▊| 493/500 [00:02<00:00, 165.97it/s]
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2026-07-17:14:56:52 INFO [_cli.run:388] Selected Tasks: ['humaneval', 'mbpp']
2
+ 2026-07-17:14:56:53 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
3
+ 2026-07-17:14:56:53 INFO [evaluator:239] Initializing local-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8420/v1/completions', 'tokenizer': 'outputs/healed/grid_general/glean_keep25_s1224/step0150', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
4
+ 2026-07-17:14:56:53 INFO [models.openai_completions:42] Remote tokenizer not supported. Using huggingface tokenizer backend.
5
+ 2026-07-17:14:56:53 INFO [models.api_models:179] Using max length 2048 - 1
6
+ 2026-07-17:14:56:53 INFO [models.api_models:200] Using tokenizer huggingface
7
+ 2026-07-17:14:56:59 INFO [evaluator_utils:446] Selected tasks:
8
+ 2026-07-17:14:56:59 INFO [evaluator_utils:480] Task: humaneval (humaneval/humaneval.yaml)
9
+ 2026-07-17:14:56:59 INFO [evaluator_utils:480] Task: mbpp (mbpp/mbpp.yaml)
10
+ 2026-07-17:14:56:59 INFO [evaluator:314] humaneval: Using gen_kwargs: {'until': ['\nclass', '\ndef', '\n#', '\nif', '\nprint'], 'max_gen_toks': 1024, 'do_sample': False}
11
+ 2026-07-17:14:56:59 INFO [evaluator:314] mbpp: Using gen_kwargs: {'until': ['[DONE]'], 'do_sample': False}
12
+ 2026-07-17:14:56:59 INFO [api.task:312] Building contexts for humaneval on rank 0...
13
+
14
  0%| | 0/164 [00:00<?, ?it/s]
15
  89%|████████▉ | 146/164 [00:00<00:00, 1454.83it/s]
16
+ 2026-07-17:14:57:00 INFO [api.task:312] Building contexts for mbpp on rank 0...
17
+
18
  0%| | 0/500 [00:00<?, ?it/s]
19
  3%|▎ | 17/500 [00:00<00:02, 162.50it/s]
20
  7%|▋ | 34/500 [00:00<00:02, 162.43it/s]
21
  10%|█ | 51/500 [00:00<00:02, 163.22it/s]
22
  14%|█▎ | 68/500 [00:00<00:02, 163.57it/s]
23
  17%|█▋ | 85/500 [00:00<00:02, 164.04it/s]
24
  20%|██ | 102/500 [00:00<00:02, 164.63it/s]
25
  24%|██▍ | 119/500 [00:00<00:02, 165.03it/s]
26
  27%|██▋ | 136/500 [00:00<00:02, 164.92it/s]
27
  31%|███ | 153/500 [00:00<00:02, 164.85it/s]
28
  34%|███▍ | 170/500 [00:01<00:01, 165.15it/s]
29
  37%|███▋ | 187/500 [00:01<00:01, 165.38it/s]
30
  41%|████ | 204/500 [00:01<00:01, 165.09it/s]
31
  44%|████▍ | 221/500 [00:01<00:01, 165.27it/s]
32
  48%|████▊ | 238/500 [00:01<00:01, 165.35it/s]
33
  51%|█████ | 255/500 [00:01<00:01, 165.52it/s]
34
  54%|█████▍ | 272/500 [00:01<00:01, 165.75it/s]
35
  58%|█████▊ | 289/500 [00:01<00:01, 165.94it/s]
36
  61%|██████ | 306/500 [00:01<00:01, 165.69it/s]
37
  65%|██████▍ | 323/500 [00:01<00:01, 165.94it/s]
38
  68%|██████▊ | 340/500 [00:02<00:00, 166.00it/s]
39
  71%|███████▏ | 357/500 [00:02<00:00, 166.04it/s]
40
  75%|███████▍ | 374/500 [00:02<00:00, 165.69it/s]
41
  78%|███████▊ | 391/500 [00:02<00:00, 165.72it/s]
42
  82%|████████▏ | 408/500 [00:02<00:00, 165.76it/s]
43
  85%|████████▌ | 425/500 [00:02<00:00, 165.87it/s]
44
  88%|████████▊ | 442/500 [00:02<00:00, 166.13it/s]
45
  92%|█████████▏| 459/500 [00:02<00:00, 166.11it/s]
46
  95%|█████████▌| 476/500 [00:02<00:00, 165.83it/s]
47
  99%|█████████▊| 493/500 [00:02<00:00, 165.97it/s]
48
+ 2026-07-17:14:57:03 INFO [evaluator:585] Running generate_until requests
49
+ 2026-07-17:14:57:03 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
50
+
51
+
52
+ 2026-07-17:15:01:00 INFO [loggers.evaluation_tracker:247] Saving results aggregated
53
+ 2026-07-17:15:01:00 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/healed/glean_keep25_s1224/student/*.jsonl
54
+ local-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8420/v1/completions', 'tokenizer': 'outputs/healed/grid_general/glean_keep25_s1224/step0150', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: None, num_fewshot: None, batch_size: 1
55
+ | Tasks |Version| Filter |n-shot| Metric | |Value | |Stderr|
56
+ |---------|------:|-----------|-----:|---------|---|-----:|---|-----:|
57
+ |humaneval| 1|create_test| 0|pass@1 |↑ |0.0183|± |0.0105|
58
+ |mbpp | 1|none | 3|pass_at_1|↑ |0.0940|± |0.0131|
59
+
evals/general_suite/healed/glean_keep25_s1224/server.log ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1224/student/results_2026-07-17T14-56-42.779609.json ADDED
@@ -0,0 +1,337 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "gsm8k_cot_zeroshot": {
4
+ "name": "gsm8k_cot_zeroshot",
5
+ "alias": "gsm8k_cot_zeroshot",
6
+ "sample_len": 1319,
7
+ "exact_match,strict-match": 0.000758150113722517,
8
+ "exact_match_stderr,strict-match": 0.0007581501137225401,
9
+ "exact_match,flexible-extract": 0.19484457922668688,
10
+ "exact_match_stderr,flexible-extract": 0.01091003940957876
11
+ },
12
+ "minerva_math500": {
13
+ "name": "minerva_math500",
14
+ "alias": "minerva_math500",
15
+ "sample_len": 500,
16
+ "exact_match,none": 0.02,
17
+ "exact_match_stderr,none": 0.006267260734501833,
18
+ "math_verify,none": 0.08,
19
+ "math_verify_stderr,none": 0.012144751540478706
20
+ },
21
+ "ifeval": {
22
+ "name": "ifeval",
23
+ "alias": "ifeval",
24
+ "sample_len": 541,
25
+ "prompt_level_strict_acc,none": 0.4343807763401109,
26
+ "prompt_level_strict_acc_stderr,none": 0.02133047365756471,
27
+ "inst_level_strict_acc,none": 0.5635491606714629,
28
+ "inst_level_strict_acc_stderr,none": "N/A",
29
+ "prompt_level_loose_acc,none": 0.4584103512014787,
30
+ "prompt_level_loose_acc_stderr,none": 0.021442010560476468,
31
+ "inst_level_loose_acc,none": 0.5875299760191847,
32
+ "inst_level_loose_acc_stderr,none": "N/A"
33
+ }
34
+ },
35
+ "group_subtasks": {},
36
+ "configs": {
37
+ "gsm8k_cot_zeroshot": {
38
+ "task": "gsm8k_cot_zeroshot",
39
+ "dataset_path": "openai/gsm8k",
40
+ "dataset_name": "main",
41
+ "training_split": "train",
42
+ "test_split": "test",
43
+ "fewshot_split": "train",
44
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
45
+ "doc_to_target": "{{answer}}",
46
+ "unsafe_code": false,
47
+ "description": "",
48
+ "target_delimiter": " ",
49
+ "fewshot_delimiter": "\n\n",
50
+ "fewshot_config": {
51
+ "sampler": "default",
52
+ "split": "train",
53
+ "process_docs": null,
54
+ "fewshot_indices": null,
55
+ "samples": null,
56
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
57
+ "doc_to_choice": null,
58
+ "doc_to_target": "{{answer}}",
59
+ "gen_prefix": null,
60
+ "fewshot_delimiter": "\n\n",
61
+ "target_delimiter": " "
62
+ },
63
+ "num_fewshot": 0,
64
+ "metric_list": [
65
+ {
66
+ "metric": "exact_match",
67
+ "aggregation": "mean",
68
+ "higher_is_better": true,
69
+ "ignore_case": true,
70
+ "ignore_punctuation": false,
71
+ "regexes_to_ignore": [
72
+ ",",
73
+ "\\$",
74
+ "(?s).*#### ",
75
+ "\\.$"
76
+ ]
77
+ }
78
+ ],
79
+ "output_type": "generate_until",
80
+ "generation_kwargs": {
81
+ "until": [
82
+ "Q:",
83
+ "</s>",
84
+ "<|im_end|>"
85
+ ],
86
+ "do_sample": false,
87
+ "max_gen_toks": 1280
88
+ },
89
+ "repeats": 1,
90
+ "filter_list": [
91
+ {
92
+ "name": "strict-match",
93
+ "filter": [
94
+ {
95
+ "function": "regex",
96
+ "regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
97
+ },
98
+ {
99
+ "function": "take_first"
100
+ }
101
+ ]
102
+ },
103
+ {
104
+ "name": "flexible-extract",
105
+ "filter": [
106
+ {
107
+ "function": "regex",
108
+ "group_select": -1,
109
+ "regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
110
+ },
111
+ {
112
+ "function": "take_first"
113
+ }
114
+ ]
115
+ }
116
+ ],
117
+ "should_decontaminate": false,
118
+ "metadata": {
119
+ "version": 3.0,
120
+ "model": "student",
121
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
122
+ "num_concurrent": 48,
123
+ "tokenized_requests": false,
124
+ "max_retries": 3,
125
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
126
+ }
127
+ },
128
+ "ifeval": {
129
+ "task": "ifeval",
130
+ "dataset_path": "google/IFEval",
131
+ "test_split": "train",
132
+ "doc_to_text": "prompt",
133
+ "doc_to_target": 0,
134
+ "unsafe_code": false,
135
+ "process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
136
+ "description": "",
137
+ "target_delimiter": " ",
138
+ "fewshot_delimiter": "\n\n",
139
+ "fewshot_config": {
140
+ "sampler": "default",
141
+ "split": null,
142
+ "process_docs": null,
143
+ "fewshot_indices": null,
144
+ "samples": null,
145
+ "doc_to_text": "prompt",
146
+ "doc_to_choice": null,
147
+ "doc_to_target": 0,
148
+ "gen_prefix": null,
149
+ "fewshot_delimiter": "\n\n",
150
+ "target_delimiter": " "
151
+ },
152
+ "num_fewshot": 0,
153
+ "metric_list": [
154
+ {
155
+ "metric": "prompt_level_strict_acc",
156
+ "aggregation": "mean",
157
+ "higher_is_better": true
158
+ },
159
+ {
160
+ "metric": "inst_level_strict_acc",
161
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
162
+ "higher_is_better": true
163
+ },
164
+ {
165
+ "metric": "prompt_level_loose_acc",
166
+ "aggregation": "mean",
167
+ "higher_is_better": true
168
+ },
169
+ {
170
+ "metric": "inst_level_loose_acc",
171
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
172
+ "higher_is_better": true
173
+ }
174
+ ],
175
+ "output_type": "generate_until",
176
+ "generation_kwargs": {
177
+ "until": [],
178
+ "do_sample": false,
179
+ "temperature": 0.0,
180
+ "max_gen_toks": 1280
181
+ },
182
+ "repeats": 1,
183
+ "should_decontaminate": false,
184
+ "metadata": {
185
+ "version": 4.0,
186
+ "model": "student",
187
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
188
+ "num_concurrent": 48,
189
+ "tokenized_requests": false,
190
+ "max_retries": 3,
191
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
192
+ }
193
+ },
194
+ "minerva_math500": {
195
+ "task": "minerva_math500",
196
+ "dataset_path": "HuggingFaceH4/MATH-500",
197
+ "dataset_name": "default",
198
+ "test_split": "test",
199
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
200
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
201
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
202
+ "unsafe_code": false,
203
+ "process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
204
+ "description": "",
205
+ "target_delimiter": " ",
206
+ "fewshot_delimiter": "\n\n",
207
+ "fewshot_config": {
208
+ "sampler": "first_n",
209
+ "split": null,
210
+ "process_docs": "<function process_docs at 0x7df0cb235b20>",
211
+ "fewshot_indices": null,
212
+ "samples": "<function list_fewshot_samples at 0x7df0cb2379c0>",
213
+ "doc_to_text": "<function doc_to_text at 0x7df0d0801800>",
214
+ "doc_to_choice": null,
215
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
216
+ "gen_prefix": null,
217
+ "fewshot_delimiter": "\n\n",
218
+ "target_delimiter": " "
219
+ },
220
+ "num_fewshot": 4,
221
+ "metric_list": [
222
+ {
223
+ "metric": "exact_match",
224
+ "aggregation": "mean",
225
+ "higher_is_better": true
226
+ },
227
+ {
228
+ "metric": "math_verify",
229
+ "aggregation": "mean",
230
+ "higher_is_better": true
231
+ }
232
+ ],
233
+ "output_type": "generate_until",
234
+ "generation_kwargs": {
235
+ "until": [
236
+ "Problem:"
237
+ ],
238
+ "do_sample": false,
239
+ "temperature": 0.0,
240
+ "max_gen_toks": 1280
241
+ },
242
+ "repeats": 1,
243
+ "should_decontaminate": false,
244
+ "metadata": {
245
+ "version": 3.0,
246
+ "model": "student",
247
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
248
+ "num_concurrent": 48,
249
+ "tokenized_requests": false,
250
+ "max_retries": 3,
251
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
252
+ }
253
+ }
254
+ },
255
+ "versions": {
256
+ "gsm8k_cot_zeroshot": 3.0,
257
+ "ifeval": 4.0,
258
+ "minerva_math500": 3.0
259
+ },
260
+ "n-shot": {
261
+ "gsm8k_cot_zeroshot": 0,
262
+ "ifeval": 0,
263
+ "minerva_math500": 4
264
+ },
265
+ "higher_is_better": {
266
+ "gsm8k_cot_zeroshot": {
267
+ "exact_match": true
268
+ },
269
+ "ifeval": {
270
+ "prompt_level_strict_acc": true,
271
+ "inst_level_strict_acc": true,
272
+ "prompt_level_loose_acc": true,
273
+ "inst_level_loose_acc": true
274
+ },
275
+ "minerva_math500": {
276
+ "exact_match": true,
277
+ "math_verify": true
278
+ }
279
+ },
280
+ "n-samples": {
281
+ "gsm8k_cot_zeroshot": {
282
+ "original": 1319,
283
+ "effective": 1319
284
+ },
285
+ "minerva_math500": {
286
+ "original": 500,
287
+ "effective": 500
288
+ },
289
+ "ifeval": {
290
+ "original": 541,
291
+ "effective": 541
292
+ }
293
+ },
294
+ "config": {
295
+ "model": "local-chat-completions",
296
+ "model_args": {
297
+ "model": "student",
298
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
299
+ "num_concurrent": 48,
300
+ "tokenized_requests": false,
301
+ "max_retries": 3
302
+ },
303
+ "batch_size": 1,
304
+ "batch_sizes": [],
305
+ "device": "cuda:0",
306
+ "use_cache": null,
307
+ "limit": null,
308
+ "bootstrap_iters": 100000,
309
+ "gen_kwargs": {
310
+ "max_gen_toks": 1280
311
+ },
312
+ "random_seed": 0,
313
+ "numpy_seed": 1234,
314
+ "torch_seed": 1234,
315
+ "fewshot_seed": 1234
316
+ },
317
+ "git_hash": "247c7f0",
318
+ "date": 1784324809.2486784,
319
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
320
+ "transformers_version": "4.57.6",
321
+ "lm_eval_version": "0.4.12",
322
+ "upper_git_hash": null,
323
+ "task_hashes": {
324
+ "gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
325
+ "minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
326
+ "ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
327
+ },
328
+ "model_source": "local-chat-completions",
329
+ "model_name": "student",
330
+ "model_name_sanitized": "student",
331
+ "system_instruction": null,
332
+ "system_instruction_sha": null,
333
+ "fewshot_as_multiturn": true,
334
+ "chat_template": "",
335
+ "chat_template_sha": null,
336
+ "total_evaluation_time_seconds": "599.8717604710109"
337
+ }
evals/general_suite/healed/glean_keep25_s1224/student/results_2026-07-17T15-01-00.225834.json ADDED
@@ -0,0 +1,225 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "humaneval": {
4
+ "name": "humaneval",
5
+ "alias": "humaneval",
6
+ "sample_len": 164,
7
+ "pass@1,create_test": 0.018292682926829267,
8
+ "pass@1_stderr,create_test": 0.010496292269168307
9
+ },
10
+ "mbpp": {
11
+ "name": "mbpp",
12
+ "alias": "mbpp",
13
+ "sample_len": 500,
14
+ "pass_at_1,none": 0.094,
15
+ "pass_at_1_stderr,none": 0.013064047561049175
16
+ }
17
+ },
18
+ "group_subtasks": {},
19
+ "configs": {
20
+ "humaneval": {
21
+ "task": "humaneval",
22
+ "dataset_path": "openai/openai_humaneval",
23
+ "test_split": "test",
24
+ "doc_to_text": "{{prompt}}",
25
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
26
+ "unsafe_code": true,
27
+ "description": "",
28
+ "target_delimiter": " ",
29
+ "fewshot_delimiter": "\n\n",
30
+ "fewshot_config": {
31
+ "sampler": "default",
32
+ "split": null,
33
+ "process_docs": null,
34
+ "fewshot_indices": null,
35
+ "samples": null,
36
+ "doc_to_text": "{{prompt}}",
37
+ "doc_to_choice": null,
38
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
39
+ "gen_prefix": null,
40
+ "fewshot_delimiter": "\n\n",
41
+ "target_delimiter": " "
42
+ },
43
+ "num_fewshot": 0,
44
+ "metric_list": [
45
+ {
46
+ "metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
47
+ "aggregation": "mean",
48
+ "higher_is_better": true,
49
+ "k": [
50
+ 1
51
+ ]
52
+ }
53
+ ],
54
+ "output_type": "generate_until",
55
+ "generation_kwargs": {
56
+ "until": [
57
+ "\nclass",
58
+ "\ndef",
59
+ "\n#",
60
+ "\nif",
61
+ "\nprint"
62
+ ],
63
+ "max_gen_toks": 1024,
64
+ "do_sample": false
65
+ },
66
+ "repeats": 1,
67
+ "filter_list": [
68
+ {
69
+ "name": "create_test",
70
+ "filter": [
71
+ {
72
+ "function": "custom",
73
+ "filter_fn": "<function build_predictions at 0x7474b627eca0>"
74
+ }
75
+ ]
76
+ }
77
+ ],
78
+ "should_decontaminate": false,
79
+ "metadata": {
80
+ "version": 1.0,
81
+ "model": "student",
82
+ "base_url": "http://127.0.0.1:8420/v1/completions",
83
+ "tokenizer": "outputs/healed/grid_general/glean_keep25_s1224/step0150",
84
+ "num_concurrent": 48,
85
+ "tokenized_requests": false,
86
+ "max_retries": 3,
87
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
88
+ }
89
+ },
90
+ "mbpp": {
91
+ "task": "mbpp",
92
+ "dataset_path": "google-research-datasets/mbpp",
93
+ "dataset_name": "full",
94
+ "test_split": "test",
95
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
96
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
97
+ "unsafe_code": true,
98
+ "description": "",
99
+ "target_delimiter": "",
100
+ "fewshot_delimiter": "\n\n",
101
+ "fewshot_config": {
102
+ "sampler": "first_n",
103
+ "split": null,
104
+ "process_docs": null,
105
+ "fewshot_indices": null,
106
+ "samples": "<function list_fewshot_samples at 0x74757f7679c0>",
107
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
108
+ "doc_to_choice": null,
109
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
110
+ "gen_prefix": null,
111
+ "fewshot_delimiter": "\n\n",
112
+ "target_delimiter": ""
113
+ },
114
+ "num_fewshot": 3,
115
+ "metric_list": [
116
+ {
117
+ "metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
118
+ "aggregation": "mean",
119
+ "higher_is_better": true
120
+ }
121
+ ],
122
+ "output_type": "generate_until",
123
+ "generation_kwargs": {
124
+ "until": [
125
+ "[DONE]"
126
+ ],
127
+ "do_sample": false
128
+ },
129
+ "repeats": 1,
130
+ "should_decontaminate": false,
131
+ "metadata": {
132
+ "version": 1.0,
133
+ "model": "student",
134
+ "base_url": "http://127.0.0.1:8420/v1/completions",
135
+ "tokenizer": "outputs/healed/grid_general/glean_keep25_s1224/step0150",
136
+ "num_concurrent": 48,
137
+ "tokenized_requests": false,
138
+ "max_retries": 3,
139
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
140
+ }
141
+ }
142
+ },
143
+ "versions": {
144
+ "humaneval": 1.0,
145
+ "mbpp": 1.0
146
+ },
147
+ "n-shot": {
148
+ "humaneval": 0,
149
+ "mbpp": 3
150
+ },
151
+ "higher_is_better": {
152
+ "humaneval": {
153
+ "pass_at_k": true,
154
+ "pass@1": true
155
+ },
156
+ "mbpp": {
157
+ "pass_at_1": true
158
+ }
159
+ },
160
+ "n-samples": {
161
+ "humaneval": {
162
+ "original": 164,
163
+ "effective": 164
164
+ },
165
+ "mbpp": {
166
+ "original": 500,
167
+ "effective": 500
168
+ }
169
+ },
170
+ "config": {
171
+ "model": "local-completions",
172
+ "model_args": {
173
+ "model": "student",
174
+ "base_url": "http://127.0.0.1:8420/v1/completions",
175
+ "tokenizer": "outputs/healed/grid_general/glean_keep25_s1224/step0150",
176
+ "num_concurrent": 48,
177
+ "tokenized_requests": false,
178
+ "max_retries": 3
179
+ },
180
+ "batch_size": 1,
181
+ "batch_sizes": [],
182
+ "device": "cuda:0",
183
+ "use_cache": null,
184
+ "limit": null,
185
+ "bootstrap_iters": 100000,
186
+ "gen_kwargs": {},
187
+ "random_seed": 0,
188
+ "numpy_seed": 1234,
189
+ "torch_seed": 1234,
190
+ "fewshot_seed": 1234
191
+ },
192
+ "git_hash": "247c7f0",
193
+ "date": 1784325412.1796849,
194
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
195
+ "transformers_version": "4.57.6",
196
+ "lm_eval_version": "0.4.12",
197
+ "upper_git_hash": null,
198
+ "tokenizer_pad_token": [
199
+ "<pad>",
200
+ "50280"
201
+ ],
202
+ "tokenizer_eos_token": [
203
+ "|||IP_ADDRESS|||",
204
+ "50279"
205
+ ],
206
+ "tokenizer_bos_token": [
207
+ "|||IP_ADDRESS|||",
208
+ "50279"
209
+ ],
210
+ "eot_token_id": 50279,
211
+ "max_length": 2047,
212
+ "task_hashes": {
213
+ "humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
214
+ "mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
215
+ },
216
+ "model_source": "local-completions",
217
+ "model_name": "student",
218
+ "model_name_sanitized": "student",
219
+ "system_instruction": null,
220
+ "system_instruction_sha": null,
221
+ "fewshot_as_multiturn": null,
222
+ "chat_template": null,
223
+ "chat_template_sha": null,
224
+ "total_evaluation_time_seconds": "254.3856012949982"
225
+ }
evals/general_suite/healed/glean_keep25_s1224/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-56-42.779609.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1224/student/samples_humaneval_2026-07-17T15-01-00.225834.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1224/student/samples_ifeval_2026-07-17T14-56-42.779609.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1224/student/samples_mbpp_2026-07-17T15-01-00.225834.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1224/student/samples_minerva_math500_2026-07-17T14-56-42.779609.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1225/student/results_2026-07-17T14-48-07.537542.json ADDED
@@ -0,0 +1,337 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "gsm8k_cot_zeroshot": {
4
+ "name": "gsm8k_cot_zeroshot",
5
+ "alias": "gsm8k_cot_zeroshot",
6
+ "sample_len": 1319,
7
+ "exact_match,strict-match": 0.0,
8
+ "exact_match_stderr,strict-match": 0.0,
9
+ "exact_match,flexible-extract": 0.21531463229719486,
10
+ "exact_match_stderr,flexible-extract": 0.01132209629457967
11
+ },
12
+ "minerva_math500": {
13
+ "name": "minerva_math500",
14
+ "alias": "minerva_math500",
15
+ "sample_len": 500,
16
+ "exact_match,none": 0.034,
17
+ "exact_match_stderr,none": 0.008112930375552172,
18
+ "math_verify,none": 0.088,
19
+ "math_verify_stderr,none": 0.012682021421471788
20
+ },
21
+ "ifeval": {
22
+ "name": "ifeval",
23
+ "alias": "ifeval",
24
+ "sample_len": 541,
25
+ "prompt_level_strict_acc,none": 0.4232902033271719,
26
+ "prompt_level_strict_acc_stderr,none": 0.021261842325248494,
27
+ "inst_level_strict_acc,none": 0.5563549160671463,
28
+ "inst_level_strict_acc_stderr,none": "N/A",
29
+ "prompt_level_loose_acc,none": 0.4565619223659889,
30
+ "prompt_level_loose_acc_stderr,none": 0.021435222545538937,
31
+ "inst_level_loose_acc,none": 0.5875299760191847,
32
+ "inst_level_loose_acc_stderr,none": "N/A"
33
+ }
34
+ },
35
+ "group_subtasks": {},
36
+ "configs": {
37
+ "gsm8k_cot_zeroshot": {
38
+ "task": "gsm8k_cot_zeroshot",
39
+ "dataset_path": "openai/gsm8k",
40
+ "dataset_name": "main",
41
+ "training_split": "train",
42
+ "test_split": "test",
43
+ "fewshot_split": "train",
44
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
45
+ "doc_to_target": "{{answer}}",
46
+ "unsafe_code": false,
47
+ "description": "",
48
+ "target_delimiter": " ",
49
+ "fewshot_delimiter": "\n\n",
50
+ "fewshot_config": {
51
+ "sampler": "default",
52
+ "split": "train",
53
+ "process_docs": null,
54
+ "fewshot_indices": null,
55
+ "samples": null,
56
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
57
+ "doc_to_choice": null,
58
+ "doc_to_target": "{{answer}}",
59
+ "gen_prefix": null,
60
+ "fewshot_delimiter": "\n\n",
61
+ "target_delimiter": " "
62
+ },
63
+ "num_fewshot": 0,
64
+ "metric_list": [
65
+ {
66
+ "metric": "exact_match",
67
+ "aggregation": "mean",
68
+ "higher_is_better": true,
69
+ "ignore_case": true,
70
+ "ignore_punctuation": false,
71
+ "regexes_to_ignore": [
72
+ ",",
73
+ "\\$",
74
+ "(?s).*#### ",
75
+ "\\.$"
76
+ ]
77
+ }
78
+ ],
79
+ "output_type": "generate_until",
80
+ "generation_kwargs": {
81
+ "until": [
82
+ "Q:",
83
+ "</s>",
84
+ "<|im_end|>"
85
+ ],
86
+ "do_sample": false,
87
+ "max_gen_toks": 1280
88
+ },
89
+ "repeats": 1,
90
+ "filter_list": [
91
+ {
92
+ "name": "strict-match",
93
+ "filter": [
94
+ {
95
+ "function": "regex",
96
+ "regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
97
+ },
98
+ {
99
+ "function": "take_first"
100
+ }
101
+ ]
102
+ },
103
+ {
104
+ "name": "flexible-extract",
105
+ "filter": [
106
+ {
107
+ "function": "regex",
108
+ "group_select": -1,
109
+ "regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
110
+ },
111
+ {
112
+ "function": "take_first"
113
+ }
114
+ ]
115
+ }
116
+ ],
117
+ "should_decontaminate": false,
118
+ "metadata": {
119
+ "version": 3.0,
120
+ "model": "student",
121
+ "base_url": "http://127.0.0.1:8421/v1/chat/completions",
122
+ "num_concurrent": 48,
123
+ "tokenized_requests": false,
124
+ "max_retries": 3,
125
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
126
+ }
127
+ },
128
+ "ifeval": {
129
+ "task": "ifeval",
130
+ "dataset_path": "google/IFEval",
131
+ "test_split": "train",
132
+ "doc_to_text": "prompt",
133
+ "doc_to_target": 0,
134
+ "unsafe_code": false,
135
+ "process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
136
+ "description": "",
137
+ "target_delimiter": " ",
138
+ "fewshot_delimiter": "\n\n",
139
+ "fewshot_config": {
140
+ "sampler": "default",
141
+ "split": null,
142
+ "process_docs": null,
143
+ "fewshot_indices": null,
144
+ "samples": null,
145
+ "doc_to_text": "prompt",
146
+ "doc_to_choice": null,
147
+ "doc_to_target": 0,
148
+ "gen_prefix": null,
149
+ "fewshot_delimiter": "\n\n",
150
+ "target_delimiter": " "
151
+ },
152
+ "num_fewshot": 0,
153
+ "metric_list": [
154
+ {
155
+ "metric": "prompt_level_strict_acc",
156
+ "aggregation": "mean",
157
+ "higher_is_better": true
158
+ },
159
+ {
160
+ "metric": "inst_level_strict_acc",
161
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
162
+ "higher_is_better": true
163
+ },
164
+ {
165
+ "metric": "prompt_level_loose_acc",
166
+ "aggregation": "mean",
167
+ "higher_is_better": true
168
+ },
169
+ {
170
+ "metric": "inst_level_loose_acc",
171
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
172
+ "higher_is_better": true
173
+ }
174
+ ],
175
+ "output_type": "generate_until",
176
+ "generation_kwargs": {
177
+ "until": [],
178
+ "do_sample": false,
179
+ "temperature": 0.0,
180
+ "max_gen_toks": 1280
181
+ },
182
+ "repeats": 1,
183
+ "should_decontaminate": false,
184
+ "metadata": {
185
+ "version": 4.0,
186
+ "model": "student",
187
+ "base_url": "http://127.0.0.1:8421/v1/chat/completions",
188
+ "num_concurrent": 48,
189
+ "tokenized_requests": false,
190
+ "max_retries": 3,
191
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
192
+ }
193
+ },
194
+ "minerva_math500": {
195
+ "task": "minerva_math500",
196
+ "dataset_path": "HuggingFaceH4/MATH-500",
197
+ "dataset_name": "default",
198
+ "test_split": "test",
199
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
200
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
201
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
202
+ "unsafe_code": false,
203
+ "process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
204
+ "description": "",
205
+ "target_delimiter": " ",
206
+ "fewshot_delimiter": "\n\n",
207
+ "fewshot_config": {
208
+ "sampler": "first_n",
209
+ "split": null,
210
+ "process_docs": "<function process_docs at 0x78d5b8035b20>",
211
+ "fewshot_indices": null,
212
+ "samples": "<function list_fewshot_samples at 0x78d5b80379c0>",
213
+ "doc_to_text": "<function doc_to_text at 0x78d5b97fd8a0>",
214
+ "doc_to_choice": null,
215
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
216
+ "gen_prefix": null,
217
+ "fewshot_delimiter": "\n\n",
218
+ "target_delimiter": " "
219
+ },
220
+ "num_fewshot": 4,
221
+ "metric_list": [
222
+ {
223
+ "metric": "exact_match",
224
+ "aggregation": "mean",
225
+ "higher_is_better": true
226
+ },
227
+ {
228
+ "metric": "math_verify",
229
+ "aggregation": "mean",
230
+ "higher_is_better": true
231
+ }
232
+ ],
233
+ "output_type": "generate_until",
234
+ "generation_kwargs": {
235
+ "until": [
236
+ "Problem:"
237
+ ],
238
+ "do_sample": false,
239
+ "temperature": 0.0,
240
+ "max_gen_toks": 1280
241
+ },
242
+ "repeats": 1,
243
+ "should_decontaminate": false,
244
+ "metadata": {
245
+ "version": 3.0,
246
+ "model": "student",
247
+ "base_url": "http://127.0.0.1:8421/v1/chat/completions",
248
+ "num_concurrent": 48,
249
+ "tokenized_requests": false,
250
+ "max_retries": 3,
251
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
252
+ }
253
+ }
254
+ },
255
+ "versions": {
256
+ "gsm8k_cot_zeroshot": 3.0,
257
+ "ifeval": 4.0,
258
+ "minerva_math500": 3.0
259
+ },
260
+ "n-shot": {
261
+ "gsm8k_cot_zeroshot": 0,
262
+ "ifeval": 0,
263
+ "minerva_math500": 4
264
+ },
265
+ "higher_is_better": {
266
+ "gsm8k_cot_zeroshot": {
267
+ "exact_match": true
268
+ },
269
+ "ifeval": {
270
+ "prompt_level_strict_acc": true,
271
+ "inst_level_strict_acc": true,
272
+ "prompt_level_loose_acc": true,
273
+ "inst_level_loose_acc": true
274
+ },
275
+ "minerva_math500": {
276
+ "exact_match": true,
277
+ "math_verify": true
278
+ }
279
+ },
280
+ "n-samples": {
281
+ "gsm8k_cot_zeroshot": {
282
+ "original": 1319,
283
+ "effective": 1319
284
+ },
285
+ "minerva_math500": {
286
+ "original": 500,
287
+ "effective": 500
288
+ },
289
+ "ifeval": {
290
+ "original": 541,
291
+ "effective": 541
292
+ }
293
+ },
294
+ "config": {
295
+ "model": "local-chat-completions",
296
+ "model_args": {
297
+ "model": "student",
298
+ "base_url": "http://127.0.0.1:8421/v1/chat/completions",
299
+ "num_concurrent": 48,
300
+ "tokenized_requests": false,
301
+ "max_retries": 3
302
+ },
303
+ "batch_size": 1,
304
+ "batch_sizes": [],
305
+ "device": "cuda:0",
306
+ "use_cache": null,
307
+ "limit": null,
308
+ "bootstrap_iters": 100000,
309
+ "gen_kwargs": {
310
+ "max_gen_toks": 1280
311
+ },
312
+ "random_seed": 0,
313
+ "numpy_seed": 1234,
314
+ "torch_seed": 1234,
315
+ "fewshot_seed": 1234
316
+ },
317
+ "git_hash": "247c7f0",
318
+ "date": 1784324273.5688035,
319
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
320
+ "transformers_version": "4.57.6",
321
+ "lm_eval_version": "0.4.12",
322
+ "upper_git_hash": null,
323
+ "task_hashes": {
324
+ "gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
325
+ "minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
326
+ "ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
327
+ },
328
+ "model_source": "local-chat-completions",
329
+ "model_name": "student",
330
+ "model_name_sanitized": "student",
331
+ "system_instruction": null,
332
+ "system_instruction_sha": null,
333
+ "fewshot_as_multiturn": true,
334
+ "chat_template": "",
335
+ "chat_template_sha": null,
336
+ "total_evaluation_time_seconds": "620.2983292259887"
337
+ }
evals/general_suite/healed/glean_keep25_s1225/student/results_2026-07-17T14-54-29.755565.json ADDED
@@ -0,0 +1,225 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "humaneval": {
4
+ "name": "humaneval",
5
+ "alias": "humaneval",
6
+ "sample_len": 164,
7
+ "pass@1,create_test": 0.018292682926829267,
8
+ "pass@1_stderr,create_test": 0.010496292269168307
9
+ },
10
+ "mbpp": {
11
+ "name": "mbpp",
12
+ "alias": "mbpp",
13
+ "sample_len": 500,
14
+ "pass_at_1,none": 0.132,
15
+ "pass_at_1_stderr,none": 0.01515292785058016
16
+ }
17
+ },
18
+ "group_subtasks": {},
19
+ "configs": {
20
+ "humaneval": {
21
+ "task": "humaneval",
22
+ "dataset_path": "openai/openai_humaneval",
23
+ "test_split": "test",
24
+ "doc_to_text": "{{prompt}}",
25
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
26
+ "unsafe_code": true,
27
+ "description": "",
28
+ "target_delimiter": " ",
29
+ "fewshot_delimiter": "\n\n",
30
+ "fewshot_config": {
31
+ "sampler": "default",
32
+ "split": null,
33
+ "process_docs": null,
34
+ "fewshot_indices": null,
35
+ "samples": null,
36
+ "doc_to_text": "{{prompt}}",
37
+ "doc_to_choice": null,
38
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
39
+ "gen_prefix": null,
40
+ "fewshot_delimiter": "\n\n",
41
+ "target_delimiter": " "
42
+ },
43
+ "num_fewshot": 0,
44
+ "metric_list": [
45
+ {
46
+ "metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
47
+ "aggregation": "mean",
48
+ "higher_is_better": true,
49
+ "k": [
50
+ 1
51
+ ]
52
+ }
53
+ ],
54
+ "output_type": "generate_until",
55
+ "generation_kwargs": {
56
+ "until": [
57
+ "\nclass",
58
+ "\ndef",
59
+ "\n#",
60
+ "\nif",
61
+ "\nprint"
62
+ ],
63
+ "max_gen_toks": 1024,
64
+ "do_sample": false
65
+ },
66
+ "repeats": 1,
67
+ "filter_list": [
68
+ {
69
+ "name": "create_test",
70
+ "filter": [
71
+ {
72
+ "function": "custom",
73
+ "filter_fn": "<function build_predictions at 0x7ceb8c426ca0>"
74
+ }
75
+ ]
76
+ }
77
+ ],
78
+ "should_decontaminate": false,
79
+ "metadata": {
80
+ "version": 1.0,
81
+ "model": "student",
82
+ "base_url": "http://127.0.0.1:8421/v1/completions",
83
+ "tokenizer": "outputs/healed/grid_general/glean_keep25_s1225/step0150",
84
+ "num_concurrent": 48,
85
+ "tokenized_requests": false,
86
+ "max_retries": 3,
87
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
88
+ }
89
+ },
90
+ "mbpp": {
91
+ "task": "mbpp",
92
+ "dataset_path": "google-research-datasets/mbpp",
93
+ "dataset_name": "full",
94
+ "test_split": "test",
95
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
96
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
97
+ "unsafe_code": true,
98
+ "description": "",
99
+ "target_delimiter": "",
100
+ "fewshot_delimiter": "\n\n",
101
+ "fewshot_config": {
102
+ "sampler": "first_n",
103
+ "split": null,
104
+ "process_docs": null,
105
+ "fewshot_indices": null,
106
+ "samples": "<function list_fewshot_samples at 0x7cec555079c0>",
107
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
108
+ "doc_to_choice": null,
109
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
110
+ "gen_prefix": null,
111
+ "fewshot_delimiter": "\n\n",
112
+ "target_delimiter": ""
113
+ },
114
+ "num_fewshot": 3,
115
+ "metric_list": [
116
+ {
117
+ "metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
118
+ "aggregation": "mean",
119
+ "higher_is_better": true
120
+ }
121
+ ],
122
+ "output_type": "generate_until",
123
+ "generation_kwargs": {
124
+ "until": [
125
+ "[DONE]"
126
+ ],
127
+ "do_sample": false
128
+ },
129
+ "repeats": 1,
130
+ "should_decontaminate": false,
131
+ "metadata": {
132
+ "version": 1.0,
133
+ "model": "student",
134
+ "base_url": "http://127.0.0.1:8421/v1/completions",
135
+ "tokenizer": "outputs/healed/grid_general/glean_keep25_s1225/step0150",
136
+ "num_concurrent": 48,
137
+ "tokenized_requests": false,
138
+ "max_retries": 3,
139
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
140
+ }
141
+ }
142
+ },
143
+ "versions": {
144
+ "humaneval": 1.0,
145
+ "mbpp": 1.0
146
+ },
147
+ "n-shot": {
148
+ "humaneval": 0,
149
+ "mbpp": 3
150
+ },
151
+ "higher_is_better": {
152
+ "humaneval": {
153
+ "pass_at_k": true,
154
+ "pass@1": true
155
+ },
156
+ "mbpp": {
157
+ "pass_at_1": true
158
+ }
159
+ },
160
+ "n-samples": {
161
+ "humaneval": {
162
+ "original": 164,
163
+ "effective": 164
164
+ },
165
+ "mbpp": {
166
+ "original": 500,
167
+ "effective": 500
168
+ }
169
+ },
170
+ "config": {
171
+ "model": "local-completions",
172
+ "model_args": {
173
+ "model": "student",
174
+ "base_url": "http://127.0.0.1:8421/v1/completions",
175
+ "tokenizer": "outputs/healed/grid_general/glean_keep25_s1225/step0150",
176
+ "num_concurrent": 48,
177
+ "tokenized_requests": false,
178
+ "max_retries": 3
179
+ },
180
+ "batch_size": 1,
181
+ "batch_sizes": [],
182
+ "device": "cuda:0",
183
+ "use_cache": null,
184
+ "limit": null,
185
+ "bootstrap_iters": 100000,
186
+ "gen_kwargs": {},
187
+ "random_seed": 0,
188
+ "numpy_seed": 1234,
189
+ "torch_seed": 1234,
190
+ "fewshot_seed": 1234
191
+ },
192
+ "git_hash": "247c7f0",
193
+ "date": 1784324897.002143,
194
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
195
+ "transformers_version": "4.57.6",
196
+ "lm_eval_version": "0.4.12",
197
+ "upper_git_hash": null,
198
+ "tokenizer_pad_token": [
199
+ "<pad>",
200
+ "50280"
201
+ ],
202
+ "tokenizer_eos_token": [
203
+ "|||IP_ADDRESS|||",
204
+ "50279"
205
+ ],
206
+ "tokenizer_bos_token": [
207
+ "|||IP_ADDRESS|||",
208
+ "50279"
209
+ ],
210
+ "eot_token_id": 50279,
211
+ "max_length": 2047,
212
+ "task_hashes": {
213
+ "humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
214
+ "mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
215
+ },
216
+ "model_source": "local-completions",
217
+ "model_name": "student",
218
+ "model_name_sanitized": "student",
219
+ "system_instruction": null,
220
+ "system_instruction_sha": null,
221
+ "fewshot_as_multiturn": null,
222
+ "chat_template": null,
223
+ "chat_template_sha": null,
224
+ "total_evaluation_time_seconds": "379.11067679699045"
225
+ }
evals/general_suite/healed/glean_keep25_s1225/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-48-07.537542.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1225/student/samples_humaneval_2026-07-17T14-54-29.755565.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1225/student/samples_ifeval_2026-07-17T14-48-07.537542.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1225/student/samples_mbpp_2026-07-17T14-54-29.755565.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1225/student/samples_minerva_math500_2026-07-17T14-48-07.537542.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1226/lm_eval.log ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1226/lm_eval_code.log ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/164 [00:00<?, ?it/s]
1
  89%|████████▉ | 146/164 [00:00<00:00, 1454.39it/s]
 
 
2
  0%| | 0/500 [00:00<?, ?it/s]
3
  3%|▎ | 17/500 [00:00<00:02, 162.30it/s]
4
  7%|▋ | 34/500 [00:00<00:02, 163.07it/s]
5
  10%|█ | 51/500 [00:00<00:02, 163.33it/s]
6
  14%|█▎ | 68/500 [00:00<00:02, 163.58it/s]
7
  17%|█▋ | 85/500 [00:00<00:02, 163.87it/s]
8
  20%|██ | 102/500 [00:00<00:02, 164.25it/s]
9
  24%|██▍ | 119/500 [00:00<00:02, 164.37it/s]
10
  27%|██▋ | 136/500 [00:00<00:02, 164.56it/s]
11
  31%|███ | 153/500 [00:00<00:02, 164.42it/s]
12
  34%|███▍ | 170/500 [00:01<00:02, 164.57it/s]
13
  37%|███▋ | 187/500 [00:01<00:01, 164.80it/s]
14
  41%|████ | 204/500 [00:01<00:01, 164.91it/s]
15
  44%|████▍ | 221/500 [00:01<00:01, 165.06it/s]
16
  48%|████▊ | 238/500 [00:01<00:01, 165.25it/s]
17
  51%|█████ | 255/500 [00:01<00:01, 165.34it/s]
18
  54%|█████▍ | 272/500 [00:01<00:01, 165.48it/s]
19
  58%|█████▊ | 289/500 [00:01<00:01, 165.48it/s]
20
  61%|██████ | 306/500 [00:01<00:01, 165.46it/s]
21
  65%|██████▍ | 323/500 [00:01<00:01, 165.51it/s]
22
  68%|██████▊ | 340/500 [00:02<00:00, 165.47it/s]
23
  71%|███████▏ | 357/500 [00:02<00:00, 165.56it/s]
24
  75%|███████▍ | 374/500 [00:02<00:00, 165.82it/s]
25
  78%|███████▊ | 391/500 [00:02<00:00, 165.80it/s]
26
  82%|████████▏ | 408/500 [00:02<00:00, 165.87it/s]
27
  85%|████████▌ | 425/500 [00:02<00:00, 165.89it/s]
28
  88%|████████▊ | 442/500 [00:02<00:00, 165.95it/s]
29
  92%|█████████▏| 459/500 [00:02<00:00, 165.85it/s]
30
  95%|█████████▌| 476/500 [00:02<00:00, 165.54it/s]
31
  99%|█████████▊| 493/500 [00:02<00:00, 165.60it/s]
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2026-07-17:14:48:43 INFO [_cli.run:388] Selected Tasks: ['humaneval', 'mbpp']
2
+ 2026-07-17:14:48:44 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
3
+ 2026-07-17:14:48:44 INFO [evaluator:239] Initializing local-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8422/v1/completions', 'tokenizer': 'outputs/healed/grid_general/glean_keep25_s1226/step0150', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
4
+ 2026-07-17:14:48:44 INFO [models.openai_completions:42] Remote tokenizer not supported. Using huggingface tokenizer backend.
5
+ 2026-07-17:14:48:44 INFO [models.api_models:179] Using max length 2048 - 1
6
+ 2026-07-17:14:48:44 INFO [models.api_models:200] Using tokenizer huggingface
7
+ 2026-07-17:14:48:51 INFO [evaluator_utils:446] Selected tasks:
8
+ 2026-07-17:14:48:51 INFO [evaluator_utils:480] Task: humaneval (humaneval/humaneval.yaml)
9
+ 2026-07-17:14:48:51 INFO [evaluator_utils:480] Task: mbpp (mbpp/mbpp.yaml)
10
+ 2026-07-17:14:48:51 INFO [evaluator:314] humaneval: Using gen_kwargs: {'until': ['\nclass', '\ndef', '\n#', '\nif', '\nprint'], 'max_gen_toks': 1024, 'do_sample': False}
11
+ 2026-07-17:14:48:51 INFO [evaluator:314] mbpp: Using gen_kwargs: {'until': ['[DONE]'], 'do_sample': False}
12
+ 2026-07-17:14:48:51 INFO [api.task:312] Building contexts for humaneval on rank 0...
13
+
14
  0%| | 0/164 [00:00<?, ?it/s]
15
  89%|████████▉ | 146/164 [00:00<00:00, 1454.39it/s]
16
+ 2026-07-17:14:48:51 INFO [api.task:312] Building contexts for mbpp on rank 0...
17
+
18
  0%| | 0/500 [00:00<?, ?it/s]
19
  3%|▎ | 17/500 [00:00<00:02, 162.30it/s]
20
  7%|▋ | 34/500 [00:00<00:02, 163.07it/s]
21
  10%|█ | 51/500 [00:00<00:02, 163.33it/s]
22
  14%|█▎ | 68/500 [00:00<00:02, 163.58it/s]
23
  17%|█▋ | 85/500 [00:00<00:02, 163.87it/s]
24
  20%|██ | 102/500 [00:00<00:02, 164.25it/s]
25
  24%|██▍ | 119/500 [00:00<00:02, 164.37it/s]
26
  27%|██▋ | 136/500 [00:00<00:02, 164.56it/s]
27
  31%|███ | 153/500 [00:00<00:02, 164.42it/s]
28
  34%|███▍ | 170/500 [00:01<00:02, 164.57it/s]
29
  37%|███▋ | 187/500 [00:01<00:01, 164.80it/s]
30
  41%|████ | 204/500 [00:01<00:01, 164.91it/s]
31
  44%|████▍ | 221/500 [00:01<00:01, 165.06it/s]
32
  48%|████▊ | 238/500 [00:01<00:01, 165.25it/s]
33
  51%|█████ | 255/500 [00:01<00:01, 165.34it/s]
34
  54%|█████▍ | 272/500 [00:01<00:01, 165.48it/s]
35
  58%|█████▊ | 289/500 [00:01<00:01, 165.48it/s]
36
  61%|██████ | 306/500 [00:01<00:01, 165.46it/s]
37
  65%|██████▍ | 323/500 [00:01<00:01, 165.51it/s]
38
  68%|██████▊ | 340/500 [00:02<00:00, 165.47it/s]
39
  71%|███████▏ | 357/500 [00:02<00:00, 165.56it/s]
40
  75%|███████▍ | 374/500 [00:02<00:00, 165.82it/s]
41
  78%|███████▊ | 391/500 [00:02<00:00, 165.80it/s]
42
  82%|████████▏ | 408/500 [00:02<00:00, 165.87it/s]
43
  85%|████████▌ | 425/500 [00:02<00:00, 165.89it/s]
44
  88%|████████▊ | 442/500 [00:02<00:00, 165.95it/s]
45
  92%|█████████▏| 459/500 [00:02<00:00, 165.85it/s]
46
  95%|█████████▌| 476/500 [00:02<00:00, 165.54it/s]
47
  99%|█████████▊| 493/500 [00:02<00:00, 165.60it/s]
48
+ 2026-07-17:14:48:54 INFO [evaluator:585] Running generate_until requests
49
+ 2026-07-17:14:48:54 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
50
+
51
+
52
+ 2026-07-17:14:54:44 INFO [loggers.evaluation_tracker:247] Saving results aggregated
53
+ 2026-07-17:14:54:44 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/healed/glean_keep25_s1226/student/*.jsonl
54
+ local-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8422/v1/completions', 'tokenizer': 'outputs/healed/grid_general/glean_keep25_s1226/step0150', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: None, num_fewshot: None, batch_size: 1
55
+ | Tasks |Version| Filter |n-shot| Metric | |Value | |Stderr|
56
+ |---------|------:|-----------|-----:|---------|---|-----:|---|-----:|
57
+ |humaneval| 1|create_test| 0|pass@1 |↑ |0.0244|± |0.0121|
58
+ |mbpp | 1|none | 3|pass_at_1|↑ |0.1100|± |0.0140|
59
+
evals/general_suite/healed/glean_keep25_s1226/server.log ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1226/student/results_2026-07-17T14-48-34.159099.json ADDED
@@ -0,0 +1,337 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "gsm8k_cot_zeroshot": {
4
+ "name": "gsm8k_cot_zeroshot",
5
+ "alias": "gsm8k_cot_zeroshot",
6
+ "sample_len": 1319,
7
+ "exact_match,strict-match": 0.001516300227445034,
8
+ "exact_match_stderr,strict-match": 0.0010717793485492636,
9
+ "exact_match,flexible-extract": 0.21228203184230476,
10
+ "exact_match_stderr,flexible-extract": 0.011263783355400308
11
+ },
12
+ "minerva_math500": {
13
+ "name": "minerva_math500",
14
+ "alias": "minerva_math500",
15
+ "sample_len": 500,
16
+ "exact_match,none": 0.03,
17
+ "exact_match_stderr,none": 0.007636532803997706,
18
+ "math_verify,none": 0.084,
19
+ "math_verify_stderr,none": 0.012417584015643749
20
+ },
21
+ "ifeval": {
22
+ "name": "ifeval",
23
+ "alias": "ifeval",
24
+ "sample_len": 541,
25
+ "prompt_level_strict_acc,none": 0.4713493530499076,
26
+ "prompt_level_strict_acc_stderr,none": 0.02148122093008049,
27
+ "inst_level_strict_acc,none": 0.60431654676259,
28
+ "inst_level_strict_acc_stderr,none": "N/A",
29
+ "prompt_level_loose_acc,none": 0.5027726432532348,
30
+ "prompt_level_loose_acc_stderr,none": 0.021516243323548144,
31
+ "inst_level_loose_acc,none": 0.6282973621103117,
32
+ "inst_level_loose_acc_stderr,none": "N/A"
33
+ }
34
+ },
35
+ "group_subtasks": {},
36
+ "configs": {
37
+ "gsm8k_cot_zeroshot": {
38
+ "task": "gsm8k_cot_zeroshot",
39
+ "dataset_path": "openai/gsm8k",
40
+ "dataset_name": "main",
41
+ "training_split": "train",
42
+ "test_split": "test",
43
+ "fewshot_split": "train",
44
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
45
+ "doc_to_target": "{{answer}}",
46
+ "unsafe_code": false,
47
+ "description": "",
48
+ "target_delimiter": " ",
49
+ "fewshot_delimiter": "\n\n",
50
+ "fewshot_config": {
51
+ "sampler": "default",
52
+ "split": "train",
53
+ "process_docs": null,
54
+ "fewshot_indices": null,
55
+ "samples": null,
56
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
57
+ "doc_to_choice": null,
58
+ "doc_to_target": "{{answer}}",
59
+ "gen_prefix": null,
60
+ "fewshot_delimiter": "\n\n",
61
+ "target_delimiter": " "
62
+ },
63
+ "num_fewshot": 0,
64
+ "metric_list": [
65
+ {
66
+ "metric": "exact_match",
67
+ "aggregation": "mean",
68
+ "higher_is_better": true,
69
+ "ignore_case": true,
70
+ "ignore_punctuation": false,
71
+ "regexes_to_ignore": [
72
+ ",",
73
+ "\\$",
74
+ "(?s).*#### ",
75
+ "\\.$"
76
+ ]
77
+ }
78
+ ],
79
+ "output_type": "generate_until",
80
+ "generation_kwargs": {
81
+ "until": [
82
+ "Q:",
83
+ "</s>",
84
+ "<|im_end|>"
85
+ ],
86
+ "do_sample": false,
87
+ "max_gen_toks": 1280
88
+ },
89
+ "repeats": 1,
90
+ "filter_list": [
91
+ {
92
+ "name": "strict-match",
93
+ "filter": [
94
+ {
95
+ "function": "regex",
96
+ "regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
97
+ },
98
+ {
99
+ "function": "take_first"
100
+ }
101
+ ]
102
+ },
103
+ {
104
+ "name": "flexible-extract",
105
+ "filter": [
106
+ {
107
+ "function": "regex",
108
+ "group_select": -1,
109
+ "regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
110
+ },
111
+ {
112
+ "function": "take_first"
113
+ }
114
+ ]
115
+ }
116
+ ],
117
+ "should_decontaminate": false,
118
+ "metadata": {
119
+ "version": 3.0,
120
+ "model": "student",
121
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
122
+ "num_concurrent": 48,
123
+ "tokenized_requests": false,
124
+ "max_retries": 3,
125
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
126
+ }
127
+ },
128
+ "ifeval": {
129
+ "task": "ifeval",
130
+ "dataset_path": "google/IFEval",
131
+ "test_split": "train",
132
+ "doc_to_text": "prompt",
133
+ "doc_to_target": 0,
134
+ "unsafe_code": false,
135
+ "process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
136
+ "description": "",
137
+ "target_delimiter": " ",
138
+ "fewshot_delimiter": "\n\n",
139
+ "fewshot_config": {
140
+ "sampler": "default",
141
+ "split": null,
142
+ "process_docs": null,
143
+ "fewshot_indices": null,
144
+ "samples": null,
145
+ "doc_to_text": "prompt",
146
+ "doc_to_choice": null,
147
+ "doc_to_target": 0,
148
+ "gen_prefix": null,
149
+ "fewshot_delimiter": "\n\n",
150
+ "target_delimiter": " "
151
+ },
152
+ "num_fewshot": 0,
153
+ "metric_list": [
154
+ {
155
+ "metric": "prompt_level_strict_acc",
156
+ "aggregation": "mean",
157
+ "higher_is_better": true
158
+ },
159
+ {
160
+ "metric": "inst_level_strict_acc",
161
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
162
+ "higher_is_better": true
163
+ },
164
+ {
165
+ "metric": "prompt_level_loose_acc",
166
+ "aggregation": "mean",
167
+ "higher_is_better": true
168
+ },
169
+ {
170
+ "metric": "inst_level_loose_acc",
171
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
172
+ "higher_is_better": true
173
+ }
174
+ ],
175
+ "output_type": "generate_until",
176
+ "generation_kwargs": {
177
+ "until": [],
178
+ "do_sample": false,
179
+ "temperature": 0.0,
180
+ "max_gen_toks": 1280
181
+ },
182
+ "repeats": 1,
183
+ "should_decontaminate": false,
184
+ "metadata": {
185
+ "version": 4.0,
186
+ "model": "student",
187
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
188
+ "num_concurrent": 48,
189
+ "tokenized_requests": false,
190
+ "max_retries": 3,
191
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
192
+ }
193
+ },
194
+ "minerva_math500": {
195
+ "task": "minerva_math500",
196
+ "dataset_path": "HuggingFaceH4/MATH-500",
197
+ "dataset_name": "default",
198
+ "test_split": "test",
199
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
200
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
201
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
202
+ "unsafe_code": false,
203
+ "process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
204
+ "description": "",
205
+ "target_delimiter": " ",
206
+ "fewshot_delimiter": "\n\n",
207
+ "fewshot_config": {
208
+ "sampler": "first_n",
209
+ "split": null,
210
+ "process_docs": "<function process_docs at 0x7b4197c31b20>",
211
+ "fewshot_indices": null,
212
+ "samples": "<function list_fewshot_samples at 0x7b4197c339c0>",
213
+ "doc_to_text": "<function doc_to_text at 0x7b419d3fd800>",
214
+ "doc_to_choice": null,
215
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
216
+ "gen_prefix": null,
217
+ "fewshot_delimiter": "\n\n",
218
+ "target_delimiter": " "
219
+ },
220
+ "num_fewshot": 4,
221
+ "metric_list": [
222
+ {
223
+ "metric": "exact_match",
224
+ "aggregation": "mean",
225
+ "higher_is_better": true
226
+ },
227
+ {
228
+ "metric": "math_verify",
229
+ "aggregation": "mean",
230
+ "higher_is_better": true
231
+ }
232
+ ],
233
+ "output_type": "generate_until",
234
+ "generation_kwargs": {
235
+ "until": [
236
+ "Problem:"
237
+ ],
238
+ "do_sample": false,
239
+ "temperature": 0.0,
240
+ "max_gen_toks": 1280
241
+ },
242
+ "repeats": 1,
243
+ "should_decontaminate": false,
244
+ "metadata": {
245
+ "version": 3.0,
246
+ "model": "student",
247
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
248
+ "num_concurrent": 48,
249
+ "tokenized_requests": false,
250
+ "max_retries": 3,
251
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
252
+ }
253
+ }
254
+ },
255
+ "versions": {
256
+ "gsm8k_cot_zeroshot": 3.0,
257
+ "ifeval": 4.0,
258
+ "minerva_math500": 3.0
259
+ },
260
+ "n-shot": {
261
+ "gsm8k_cot_zeroshot": 0,
262
+ "ifeval": 0,
263
+ "minerva_math500": 4
264
+ },
265
+ "higher_is_better": {
266
+ "gsm8k_cot_zeroshot": {
267
+ "exact_match": true
268
+ },
269
+ "ifeval": {
270
+ "prompt_level_strict_acc": true,
271
+ "inst_level_strict_acc": true,
272
+ "prompt_level_loose_acc": true,
273
+ "inst_level_loose_acc": true
274
+ },
275
+ "minerva_math500": {
276
+ "exact_match": true,
277
+ "math_verify": true
278
+ }
279
+ },
280
+ "n-samples": {
281
+ "gsm8k_cot_zeroshot": {
282
+ "original": 1319,
283
+ "effective": 1319
284
+ },
285
+ "minerva_math500": {
286
+ "original": 500,
287
+ "effective": 500
288
+ },
289
+ "ifeval": {
290
+ "original": 541,
291
+ "effective": 541
292
+ }
293
+ },
294
+ "config": {
295
+ "model": "local-chat-completions",
296
+ "model_args": {
297
+ "model": "student",
298
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
299
+ "num_concurrent": 48,
300
+ "tokenized_requests": false,
301
+ "max_retries": 3
302
+ },
303
+ "batch_size": 1,
304
+ "batch_sizes": [],
305
+ "device": "cuda:0",
306
+ "use_cache": null,
307
+ "limit": null,
308
+ "bootstrap_iters": 100000,
309
+ "gen_kwargs": {
310
+ "max_gen_toks": 1280
311
+ },
312
+ "random_seed": 0,
313
+ "numpy_seed": 1234,
314
+ "torch_seed": 1234,
315
+ "fewshot_seed": 1234
316
+ },
317
+ "git_hash": "247c7f0",
318
+ "date": 1784324361.384908,
319
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
320
+ "transformers_version": "4.57.6",
321
+ "lm_eval_version": "0.4.12",
322
+ "upper_git_hash": null,
323
+ "task_hashes": {
324
+ "gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
325
+ "minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
326
+ "ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
327
+ },
328
+ "model_source": "local-chat-completions",
329
+ "model_name": "student",
330
+ "model_name_sanitized": "student",
331
+ "system_instruction": null,
332
+ "system_instruction_sha": null,
333
+ "fewshot_as_multiturn": true,
334
+ "chat_template": "",
335
+ "chat_template_sha": null,
336
+ "total_evaluation_time_seconds": "559.1113323710015"
337
+ }
evals/general_suite/healed/glean_keep25_s1226/student/results_2026-07-17T14-54-44.918246.json ADDED
@@ -0,0 +1,225 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "humaneval": {
4
+ "name": "humaneval",
5
+ "alias": "humaneval",
6
+ "sample_len": 164,
7
+ "pass@1,create_test": 0.024390243902439025,
8
+ "pass@1_stderr,create_test": 0.01208237572239194
9
+ },
10
+ "mbpp": {
11
+ "name": "mbpp",
12
+ "alias": "mbpp",
13
+ "sample_len": 500,
14
+ "pass_at_1,none": 0.11,
15
+ "pass_at_1_stderr,none": 0.014006869199415602
16
+ }
17
+ },
18
+ "group_subtasks": {},
19
+ "configs": {
20
+ "humaneval": {
21
+ "task": "humaneval",
22
+ "dataset_path": "openai/openai_humaneval",
23
+ "test_split": "test",
24
+ "doc_to_text": "{{prompt}}",
25
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
26
+ "unsafe_code": true,
27
+ "description": "",
28
+ "target_delimiter": " ",
29
+ "fewshot_delimiter": "\n\n",
30
+ "fewshot_config": {
31
+ "sampler": "default",
32
+ "split": null,
33
+ "process_docs": null,
34
+ "fewshot_indices": null,
35
+ "samples": null,
36
+ "doc_to_text": "{{prompt}}",
37
+ "doc_to_choice": null,
38
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
39
+ "gen_prefix": null,
40
+ "fewshot_delimiter": "\n\n",
41
+ "target_delimiter": " "
42
+ },
43
+ "num_fewshot": 0,
44
+ "metric_list": [
45
+ {
46
+ "metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
47
+ "aggregation": "mean",
48
+ "higher_is_better": true,
49
+ "k": [
50
+ 1
51
+ ]
52
+ }
53
+ ],
54
+ "output_type": "generate_until",
55
+ "generation_kwargs": {
56
+ "until": [
57
+ "\nclass",
58
+ "\ndef",
59
+ "\n#",
60
+ "\nif",
61
+ "\nprint"
62
+ ],
63
+ "max_gen_toks": 1024,
64
+ "do_sample": false
65
+ },
66
+ "repeats": 1,
67
+ "filter_list": [
68
+ {
69
+ "name": "create_test",
70
+ "filter": [
71
+ {
72
+ "function": "custom",
73
+ "filter_fn": "<function build_predictions at 0x789d7586aca0>"
74
+ }
75
+ ]
76
+ }
77
+ ],
78
+ "should_decontaminate": false,
79
+ "metadata": {
80
+ "version": 1.0,
81
+ "model": "student",
82
+ "base_url": "http://127.0.0.1:8422/v1/completions",
83
+ "tokenizer": "outputs/healed/grid_general/glean_keep25_s1226/step0150",
84
+ "num_concurrent": 48,
85
+ "tokenized_requests": false,
86
+ "max_retries": 3,
87
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
88
+ }
89
+ },
90
+ "mbpp": {
91
+ "task": "mbpp",
92
+ "dataset_path": "google-research-datasets/mbpp",
93
+ "dataset_name": "full",
94
+ "test_split": "test",
95
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
96
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
97
+ "unsafe_code": true,
98
+ "description": "",
99
+ "target_delimiter": "",
100
+ "fewshot_delimiter": "\n\n",
101
+ "fewshot_config": {
102
+ "sampler": "first_n",
103
+ "split": null,
104
+ "process_docs": null,
105
+ "fewshot_indices": null,
106
+ "samples": "<function list_fewshot_samples at 0x789e3e90b9c0>",
107
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
108
+ "doc_to_choice": null,
109
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
110
+ "gen_prefix": null,
111
+ "fewshot_delimiter": "\n\n",
112
+ "target_delimiter": ""
113
+ },
114
+ "num_fewshot": 3,
115
+ "metric_list": [
116
+ {
117
+ "metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
118
+ "aggregation": "mean",
119
+ "higher_is_better": true
120
+ }
121
+ ],
122
+ "output_type": "generate_until",
123
+ "generation_kwargs": {
124
+ "until": [
125
+ "[DONE]"
126
+ ],
127
+ "do_sample": false
128
+ },
129
+ "repeats": 1,
130
+ "should_decontaminate": false,
131
+ "metadata": {
132
+ "version": 1.0,
133
+ "model": "student",
134
+ "base_url": "http://127.0.0.1:8422/v1/completions",
135
+ "tokenizer": "outputs/healed/grid_general/glean_keep25_s1226/step0150",
136
+ "num_concurrent": 48,
137
+ "tokenized_requests": false,
138
+ "max_retries": 3,
139
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
140
+ }
141
+ }
142
+ },
143
+ "versions": {
144
+ "humaneval": 1.0,
145
+ "mbpp": 1.0
146
+ },
147
+ "n-shot": {
148
+ "humaneval": 0,
149
+ "mbpp": 3
150
+ },
151
+ "higher_is_better": {
152
+ "humaneval": {
153
+ "pass_at_k": true,
154
+ "pass@1": true
155
+ },
156
+ "mbpp": {
157
+ "pass_at_1": true
158
+ }
159
+ },
160
+ "n-samples": {
161
+ "humaneval": {
162
+ "original": 164,
163
+ "effective": 164
164
+ },
165
+ "mbpp": {
166
+ "original": 500,
167
+ "effective": 500
168
+ }
169
+ },
170
+ "config": {
171
+ "model": "local-completions",
172
+ "model_args": {
173
+ "model": "student",
174
+ "base_url": "http://127.0.0.1:8422/v1/completions",
175
+ "tokenizer": "outputs/healed/grid_general/glean_keep25_s1226/step0150",
176
+ "num_concurrent": 48,
177
+ "tokenized_requests": false,
178
+ "max_retries": 3
179
+ },
180
+ "batch_size": 1,
181
+ "batch_sizes": [],
182
+ "device": "cuda:0",
183
+ "use_cache": null,
184
+ "limit": null,
185
+ "bootstrap_iters": 100000,
186
+ "gen_kwargs": {},
187
+ "random_seed": 0,
188
+ "numpy_seed": 1234,
189
+ "torch_seed": 1234,
190
+ "fewshot_seed": 1234
191
+ },
192
+ "git_hash": "247c7f0",
193
+ "date": 1784324923.614639,
194
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.23\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
195
+ "transformers_version": "4.57.6",
196
+ "lm_eval_version": "0.4.12",
197
+ "upper_git_hash": null,
198
+ "tokenizer_pad_token": [
199
+ "<pad>",
200
+ "50280"
201
+ ],
202
+ "tokenizer_eos_token": [
203
+ "|||IP_ADDRESS|||",
204
+ "50279"
205
+ ],
206
+ "tokenizer_bos_token": [
207
+ "|||IP_ADDRESS|||",
208
+ "50279"
209
+ ],
210
+ "eot_token_id": 50279,
211
+ "max_length": 2047,
212
+ "task_hashes": {
213
+ "humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
214
+ "mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
215
+ },
216
+ "model_source": "local-completions",
217
+ "model_name": "student",
218
+ "model_name_sanitized": "student",
219
+ "system_instruction": null,
220
+ "system_instruction_sha": null,
221
+ "fewshot_as_multiturn": null,
222
+ "chat_template": null,
223
+ "chat_template_sha": null,
224
+ "total_evaluation_time_seconds": "367.6433539800055"
225
+ }
evals/general_suite/healed/glean_keep25_s1226/student/samples_gsm8k_cot_zeroshot_2026-07-17T14-48-34.159099.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1226/student/samples_humaneval_2026-07-17T14-54-44.918246.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1226/student/samples_ifeval_2026-07-17T14-48-34.159099.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1226/student/samples_mbpp_2026-07-17T14-54-44.918246.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep25_s1226/student/samples_minerva_math500_2026-07-17T14-48-34.159099.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224/student/results_2026-07-18T10-11-54.989656.json ADDED
@@ -0,0 +1,337 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "gsm8k_cot_zeroshot": {
4
+ "name": "gsm8k_cot_zeroshot",
5
+ "alias": "gsm8k_cot_zeroshot",
6
+ "sample_len": 1319,
7
+ "exact_match,strict-match": 0.000758150113722517,
8
+ "exact_match_stderr,strict-match": 0.0007581501137225239,
9
+ "exact_match,flexible-extract": 0.5572403335860501,
10
+ "exact_match_stderr,flexible-extract": 0.013681937191764623
11
+ },
12
+ "minerva_math500": {
13
+ "name": "minerva_math500",
14
+ "alias": "minerva_math500",
15
+ "sample_len": 500,
16
+ "exact_match,none": 0.118,
17
+ "exact_match_stderr,none": 0.014441922942480794,
18
+ "math_verify,none": 0.186,
19
+ "math_verify_stderr,none": 0.01741880678058399
20
+ },
21
+ "ifeval": {
22
+ "name": "ifeval",
23
+ "alias": "ifeval",
24
+ "sample_len": 541,
25
+ "prompt_level_strict_acc,none": 0.5582255083179297,
26
+ "prompt_level_strict_acc_stderr,none": 0.02137018475895099,
27
+ "inst_level_strict_acc,none": 0.6762589928057554,
28
+ "inst_level_strict_acc_stderr,none": "N/A",
29
+ "prompt_level_loose_acc,none": 0.5914972273567468,
30
+ "prompt_level_loose_acc_stderr,none": 0.021153244098720447,
31
+ "inst_level_loose_acc,none": 0.7086330935251799,
32
+ "inst_level_loose_acc_stderr,none": "N/A"
33
+ }
34
+ },
35
+ "group_subtasks": {},
36
+ "configs": {
37
+ "gsm8k_cot_zeroshot": {
38
+ "task": "gsm8k_cot_zeroshot",
39
+ "dataset_path": "openai/gsm8k",
40
+ "dataset_name": "main",
41
+ "training_split": "train",
42
+ "test_split": "test",
43
+ "fewshot_split": "train",
44
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
45
+ "doc_to_target": "{{answer}}",
46
+ "unsafe_code": false,
47
+ "description": "",
48
+ "target_delimiter": " ",
49
+ "fewshot_delimiter": "\n\n",
50
+ "fewshot_config": {
51
+ "sampler": "default",
52
+ "split": "train",
53
+ "process_docs": null,
54
+ "fewshot_indices": null,
55
+ "samples": null,
56
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
57
+ "doc_to_choice": null,
58
+ "doc_to_target": "{{answer}}",
59
+ "gen_prefix": null,
60
+ "fewshot_delimiter": "\n\n",
61
+ "target_delimiter": " "
62
+ },
63
+ "num_fewshot": 0,
64
+ "metric_list": [
65
+ {
66
+ "metric": "exact_match",
67
+ "aggregation": "mean",
68
+ "higher_is_better": true,
69
+ "ignore_case": true,
70
+ "ignore_punctuation": false,
71
+ "regexes_to_ignore": [
72
+ ",",
73
+ "\\$",
74
+ "(?s).*#### ",
75
+ "\\.$"
76
+ ]
77
+ }
78
+ ],
79
+ "output_type": "generate_until",
80
+ "generation_kwargs": {
81
+ "until": [
82
+ "Q:",
83
+ "</s>",
84
+ "<|im_end|>"
85
+ ],
86
+ "do_sample": false,
87
+ "max_gen_toks": 1280
88
+ },
89
+ "repeats": 1,
90
+ "filter_list": [
91
+ {
92
+ "name": "strict-match",
93
+ "filter": [
94
+ {
95
+ "function": "regex",
96
+ "regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
97
+ },
98
+ {
99
+ "function": "take_first"
100
+ }
101
+ ]
102
+ },
103
+ {
104
+ "name": "flexible-extract",
105
+ "filter": [
106
+ {
107
+ "function": "regex",
108
+ "group_select": -1,
109
+ "regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
110
+ },
111
+ {
112
+ "function": "take_first"
113
+ }
114
+ ]
115
+ }
116
+ ],
117
+ "should_decontaminate": false,
118
+ "metadata": {
119
+ "version": 3.0,
120
+ "model": "student",
121
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
122
+ "num_concurrent": 48,
123
+ "tokenized_requests": false,
124
+ "max_retries": 3,
125
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
126
+ }
127
+ },
128
+ "ifeval": {
129
+ "task": "ifeval",
130
+ "dataset_path": "google/IFEval",
131
+ "test_split": "train",
132
+ "doc_to_text": "prompt",
133
+ "doc_to_target": 0,
134
+ "unsafe_code": false,
135
+ "process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
136
+ "description": "",
137
+ "target_delimiter": " ",
138
+ "fewshot_delimiter": "\n\n",
139
+ "fewshot_config": {
140
+ "sampler": "default",
141
+ "split": null,
142
+ "process_docs": null,
143
+ "fewshot_indices": null,
144
+ "samples": null,
145
+ "doc_to_text": "prompt",
146
+ "doc_to_choice": null,
147
+ "doc_to_target": 0,
148
+ "gen_prefix": null,
149
+ "fewshot_delimiter": "\n\n",
150
+ "target_delimiter": " "
151
+ },
152
+ "num_fewshot": 0,
153
+ "metric_list": [
154
+ {
155
+ "metric": "prompt_level_strict_acc",
156
+ "aggregation": "mean",
157
+ "higher_is_better": true
158
+ },
159
+ {
160
+ "metric": "inst_level_strict_acc",
161
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
162
+ "higher_is_better": true
163
+ },
164
+ {
165
+ "metric": "prompt_level_loose_acc",
166
+ "aggregation": "mean",
167
+ "higher_is_better": true
168
+ },
169
+ {
170
+ "metric": "inst_level_loose_acc",
171
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
172
+ "higher_is_better": true
173
+ }
174
+ ],
175
+ "output_type": "generate_until",
176
+ "generation_kwargs": {
177
+ "until": [],
178
+ "do_sample": false,
179
+ "temperature": 0.0,
180
+ "max_gen_toks": 1280
181
+ },
182
+ "repeats": 1,
183
+ "should_decontaminate": false,
184
+ "metadata": {
185
+ "version": 4.0,
186
+ "model": "student",
187
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
188
+ "num_concurrent": 48,
189
+ "tokenized_requests": false,
190
+ "max_retries": 3,
191
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
192
+ }
193
+ },
194
+ "minerva_math500": {
195
+ "task": "minerva_math500",
196
+ "dataset_path": "HuggingFaceH4/MATH-500",
197
+ "dataset_name": "default",
198
+ "test_split": "test",
199
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
200
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
201
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
202
+ "unsafe_code": false,
203
+ "process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
204
+ "description": "",
205
+ "target_delimiter": " ",
206
+ "fewshot_delimiter": "\n\n",
207
+ "fewshot_config": {
208
+ "sampler": "first_n",
209
+ "split": null,
210
+ "process_docs": "<function process_docs at 0x704287c25b20>",
211
+ "fewshot_indices": null,
212
+ "samples": "<function list_fewshot_samples at 0x704287c279c0>",
213
+ "doc_to_text": "<function doc_to_text at 0x70428d3f18a0>",
214
+ "doc_to_choice": null,
215
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
216
+ "gen_prefix": null,
217
+ "fewshot_delimiter": "\n\n",
218
+ "target_delimiter": " "
219
+ },
220
+ "num_fewshot": 4,
221
+ "metric_list": [
222
+ {
223
+ "metric": "exact_match",
224
+ "aggregation": "mean",
225
+ "higher_is_better": true
226
+ },
227
+ {
228
+ "metric": "math_verify",
229
+ "aggregation": "mean",
230
+ "higher_is_better": true
231
+ }
232
+ ],
233
+ "output_type": "generate_until",
234
+ "generation_kwargs": {
235
+ "until": [
236
+ "Problem:"
237
+ ],
238
+ "do_sample": false,
239
+ "temperature": 0.0,
240
+ "max_gen_toks": 1280
241
+ },
242
+ "repeats": 1,
243
+ "should_decontaminate": false,
244
+ "metadata": {
245
+ "version": 3.0,
246
+ "model": "student",
247
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
248
+ "num_concurrent": 48,
249
+ "tokenized_requests": false,
250
+ "max_retries": 3,
251
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
252
+ }
253
+ }
254
+ },
255
+ "versions": {
256
+ "gsm8k_cot_zeroshot": 3.0,
257
+ "ifeval": 4.0,
258
+ "minerva_math500": 3.0
259
+ },
260
+ "n-shot": {
261
+ "gsm8k_cot_zeroshot": 0,
262
+ "ifeval": 0,
263
+ "minerva_math500": 4
264
+ },
265
+ "higher_is_better": {
266
+ "gsm8k_cot_zeroshot": {
267
+ "exact_match": true
268
+ },
269
+ "ifeval": {
270
+ "prompt_level_strict_acc": true,
271
+ "inst_level_strict_acc": true,
272
+ "prompt_level_loose_acc": true,
273
+ "inst_level_loose_acc": true
274
+ },
275
+ "minerva_math500": {
276
+ "exact_match": true,
277
+ "math_verify": true
278
+ }
279
+ },
280
+ "n-samples": {
281
+ "gsm8k_cot_zeroshot": {
282
+ "original": 1319,
283
+ "effective": 1319
284
+ },
285
+ "minerva_math500": {
286
+ "original": 500,
287
+ "effective": 500
288
+ },
289
+ "ifeval": {
290
+ "original": 541,
291
+ "effective": 541
292
+ }
293
+ },
294
+ "config": {
295
+ "model": "local-chat-completions",
296
+ "model_args": {
297
+ "model": "student",
298
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
299
+ "num_concurrent": 48,
300
+ "tokenized_requests": false,
301
+ "max_retries": 3
302
+ },
303
+ "batch_size": 1,
304
+ "batch_sizes": [],
305
+ "device": "cuda:0",
306
+ "use_cache": null,
307
+ "limit": null,
308
+ "bootstrap_iters": 100000,
309
+ "gen_kwargs": {
310
+ "max_gen_toks": 1280
311
+ },
312
+ "random_seed": 0,
313
+ "numpy_seed": 1234,
314
+ "torch_seed": 1234,
315
+ "fewshot_seed": 1234
316
+ },
317
+ "git_hash": "4cd474a",
318
+ "date": 1784394159.0160384,
319
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
320
+ "transformers_version": "4.57.6",
321
+ "lm_eval_version": "0.4.12",
322
+ "upper_git_hash": null,
323
+ "task_hashes": {
324
+ "gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
325
+ "minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
326
+ "ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
327
+ },
328
+ "model_source": "local-chat-completions",
329
+ "model_name": "student",
330
+ "model_name_sanitized": "student",
331
+ "system_instruction": null,
332
+ "system_instruction_sha": null,
333
+ "fewshot_as_multiturn": true,
334
+ "chat_template": "",
335
+ "chat_template_sha": null,
336
+ "total_evaluation_time_seconds": "562.3516382400012"
337
+ }
evals/general_suite/healed/glean_keep50_s1224/student/results_2026-07-18T10-15-21.380553.json ADDED
@@ -0,0 +1,225 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "humaneval": {
4
+ "name": "humaneval",
5
+ "alias": "humaneval",
6
+ "sample_len": 164,
7
+ "pass@1,create_test": 0.2682926829268293,
8
+ "pass@1_stderr,create_test": 0.03470398212814533
9
+ },
10
+ "mbpp": {
11
+ "name": "mbpp",
12
+ "alias": "mbpp",
13
+ "sample_len": 500,
14
+ "pass_at_1,none": 0.23,
15
+ "pass_at_1_stderr,none": 0.018839050391123137
16
+ }
17
+ },
18
+ "group_subtasks": {},
19
+ "configs": {
20
+ "humaneval": {
21
+ "task": "humaneval",
22
+ "dataset_path": "openai/openai_humaneval",
23
+ "test_split": "test",
24
+ "doc_to_text": "{{prompt}}",
25
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
26
+ "unsafe_code": true,
27
+ "description": "",
28
+ "target_delimiter": " ",
29
+ "fewshot_delimiter": "\n\n",
30
+ "fewshot_config": {
31
+ "sampler": "default",
32
+ "split": null,
33
+ "process_docs": null,
34
+ "fewshot_indices": null,
35
+ "samples": null,
36
+ "doc_to_text": "{{prompt}}",
37
+ "doc_to_choice": null,
38
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
39
+ "gen_prefix": null,
40
+ "fewshot_delimiter": "\n\n",
41
+ "target_delimiter": " "
42
+ },
43
+ "num_fewshot": 0,
44
+ "metric_list": [
45
+ {
46
+ "metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
47
+ "aggregation": "mean",
48
+ "higher_is_better": true,
49
+ "k": [
50
+ 1
51
+ ]
52
+ }
53
+ ],
54
+ "output_type": "generate_until",
55
+ "generation_kwargs": {
56
+ "until": [
57
+ "\nclass",
58
+ "\ndef",
59
+ "\n#",
60
+ "\nif",
61
+ "\nprint"
62
+ ],
63
+ "max_gen_toks": 1024,
64
+ "do_sample": false
65
+ },
66
+ "repeats": 1,
67
+ "filter_list": [
68
+ {
69
+ "name": "create_test",
70
+ "filter": [
71
+ {
72
+ "function": "custom",
73
+ "filter_fn": "<function build_predictions at 0x7d20f9872ca0>"
74
+ }
75
+ ]
76
+ }
77
+ ],
78
+ "should_decontaminate": false,
79
+ "metadata": {
80
+ "version": 1.0,
81
+ "model": "student",
82
+ "base_url": "http://127.0.0.1:8420/v1/completions",
83
+ "tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0150",
84
+ "num_concurrent": 48,
85
+ "tokenized_requests": false,
86
+ "max_retries": 3,
87
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
88
+ }
89
+ },
90
+ "mbpp": {
91
+ "task": "mbpp",
92
+ "dataset_path": "google-research-datasets/mbpp",
93
+ "dataset_name": "full",
94
+ "test_split": "test",
95
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
96
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
97
+ "unsafe_code": true,
98
+ "description": "",
99
+ "target_delimiter": "",
100
+ "fewshot_delimiter": "\n\n",
101
+ "fewshot_config": {
102
+ "sampler": "first_n",
103
+ "split": null,
104
+ "process_docs": null,
105
+ "fewshot_indices": null,
106
+ "samples": "<function list_fewshot_samples at 0x7d21c2b1f9c0>",
107
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
108
+ "doc_to_choice": null,
109
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
110
+ "gen_prefix": null,
111
+ "fewshot_delimiter": "\n\n",
112
+ "target_delimiter": ""
113
+ },
114
+ "num_fewshot": 3,
115
+ "metric_list": [
116
+ {
117
+ "metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
118
+ "aggregation": "mean",
119
+ "higher_is_better": true
120
+ }
121
+ ],
122
+ "output_type": "generate_until",
123
+ "generation_kwargs": {
124
+ "until": [
125
+ "[DONE]"
126
+ ],
127
+ "do_sample": false
128
+ },
129
+ "repeats": 1,
130
+ "should_decontaminate": false,
131
+ "metadata": {
132
+ "version": 1.0,
133
+ "model": "student",
134
+ "base_url": "http://127.0.0.1:8420/v1/completions",
135
+ "tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0150",
136
+ "num_concurrent": 48,
137
+ "tokenized_requests": false,
138
+ "max_retries": 3,
139
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
140
+ }
141
+ }
142
+ },
143
+ "versions": {
144
+ "humaneval": 1.0,
145
+ "mbpp": 1.0
146
+ },
147
+ "n-shot": {
148
+ "humaneval": 0,
149
+ "mbpp": 3
150
+ },
151
+ "higher_is_better": {
152
+ "humaneval": {
153
+ "pass_at_k": true,
154
+ "pass@1": true
155
+ },
156
+ "mbpp": {
157
+ "pass_at_1": true
158
+ }
159
+ },
160
+ "n-samples": {
161
+ "humaneval": {
162
+ "original": 164,
163
+ "effective": 164
164
+ },
165
+ "mbpp": {
166
+ "original": 500,
167
+ "effective": 500
168
+ }
169
+ },
170
+ "config": {
171
+ "model": "local-completions",
172
+ "model_args": {
173
+ "model": "student",
174
+ "base_url": "http://127.0.0.1:8420/v1/completions",
175
+ "tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0150",
176
+ "num_concurrent": 48,
177
+ "tokenized_requests": false,
178
+ "max_retries": 3
179
+ },
180
+ "batch_size": 1,
181
+ "batch_sizes": [],
182
+ "device": "cuda:0",
183
+ "use_cache": null,
184
+ "limit": null,
185
+ "bootstrap_iters": 100000,
186
+ "gen_kwargs": {},
187
+ "random_seed": 0,
188
+ "numpy_seed": 1234,
189
+ "torch_seed": 1234,
190
+ "fewshot_seed": 1234
191
+ },
192
+ "git_hash": "4cd474a",
193
+ "date": 1784394724.0320532,
194
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
195
+ "transformers_version": "4.57.6",
196
+ "lm_eval_version": "0.4.12",
197
+ "upper_git_hash": null,
198
+ "tokenizer_pad_token": [
199
+ "<pad>",
200
+ "50280"
201
+ ],
202
+ "tokenizer_eos_token": [
203
+ "|||IP_ADDRESS|||",
204
+ "50279"
205
+ ],
206
+ "tokenizer_bos_token": [
207
+ "|||IP_ADDRESS|||",
208
+ "50279"
209
+ ],
210
+ "eot_token_id": 50279,
211
+ "max_length": 2047,
212
+ "task_hashes": {
213
+ "humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
214
+ "mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
215
+ },
216
+ "model_source": "local-completions",
217
+ "model_name": "student",
218
+ "model_name_sanitized": "student",
219
+ "system_instruction": null,
220
+ "system_instruction_sha": null,
221
+ "fewshot_as_multiturn": null,
222
+ "chat_template": null,
223
+ "chat_template_sha": null,
224
+ "total_evaluation_time_seconds": "203.6669814880006"
225
+ }
evals/general_suite/healed/glean_keep50_s1224/student/samples_gsm8k_cot_zeroshot_2026-07-18T10-11-54.989656.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224/student/samples_humaneval_2026-07-18T10-15-21.380553.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224/student/samples_ifeval_2026-07-18T10-11-54.989656.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224/student/samples_mbpp_2026-07-18T10-15-21.380553.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224/student/samples_minerva_math500_2026-07-18T10-11-54.989656.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/results_2026-07-19T21-36-45.489848.json ADDED
@@ -0,0 +1,337 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "gsm8k_cot_zeroshot": {
4
+ "name": "gsm8k_cot_zeroshot",
5
+ "alias": "gsm8k_cot_zeroshot",
6
+ "sample_len": 1319,
7
+ "exact_match,strict-match": 0.0,
8
+ "exact_match_stderr,strict-match": 0.0,
9
+ "exact_match,flexible-extract": 0.5526914329037149,
10
+ "exact_match_stderr,flexible-extract": 0.013695795709089898
11
+ },
12
+ "minerva_math500": {
13
+ "name": "minerva_math500",
14
+ "alias": "minerva_math500",
15
+ "sample_len": 500,
16
+ "exact_match,none": 0.118,
17
+ "exact_match_stderr,none": 0.014441922942480794,
18
+ "math_verify,none": 0.21,
19
+ "math_verify_stderr,none": 0.01823362086530592
20
+ },
21
+ "ifeval": {
22
+ "name": "ifeval",
23
+ "alias": "ifeval",
24
+ "sample_len": 541,
25
+ "prompt_level_strict_acc,none": 0.5415896487985212,
26
+ "prompt_level_strict_acc_stderr,none": 0.021442010560476468,
27
+ "inst_level_strict_acc,none": 0.6630695443645084,
28
+ "inst_level_strict_acc_stderr,none": "N/A",
29
+ "prompt_level_loose_acc,none": 0.5656192236598891,
30
+ "prompt_level_loose_acc_stderr,none": 0.021330473657564707,
31
+ "inst_level_loose_acc,none": 0.6906474820143885,
32
+ "inst_level_loose_acc_stderr,none": "N/A"
33
+ }
34
+ },
35
+ "group_subtasks": {},
36
+ "configs": {
37
+ "gsm8k_cot_zeroshot": {
38
+ "task": "gsm8k_cot_zeroshot",
39
+ "dataset_path": "openai/gsm8k",
40
+ "dataset_name": "main",
41
+ "training_split": "train",
42
+ "test_split": "test",
43
+ "fewshot_split": "train",
44
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
45
+ "doc_to_target": "{{answer}}",
46
+ "unsafe_code": false,
47
+ "description": "",
48
+ "target_delimiter": " ",
49
+ "fewshot_delimiter": "\n\n",
50
+ "fewshot_config": {
51
+ "sampler": "default",
52
+ "split": "train",
53
+ "process_docs": null,
54
+ "fewshot_indices": null,
55
+ "samples": null,
56
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
57
+ "doc_to_choice": null,
58
+ "doc_to_target": "{{answer}}",
59
+ "gen_prefix": null,
60
+ "fewshot_delimiter": "\n\n",
61
+ "target_delimiter": " "
62
+ },
63
+ "num_fewshot": 0,
64
+ "metric_list": [
65
+ {
66
+ "metric": "exact_match",
67
+ "aggregation": "mean",
68
+ "higher_is_better": true,
69
+ "ignore_case": true,
70
+ "ignore_punctuation": false,
71
+ "regexes_to_ignore": [
72
+ ",",
73
+ "\\$",
74
+ "(?s).*#### ",
75
+ "\\.$"
76
+ ]
77
+ }
78
+ ],
79
+ "output_type": "generate_until",
80
+ "generation_kwargs": {
81
+ "until": [
82
+ "Q:",
83
+ "</s>",
84
+ "<|im_end|>"
85
+ ],
86
+ "do_sample": false,
87
+ "max_gen_toks": 1280
88
+ },
89
+ "repeats": 1,
90
+ "filter_list": [
91
+ {
92
+ "name": "strict-match",
93
+ "filter": [
94
+ {
95
+ "function": "regex",
96
+ "regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
97
+ },
98
+ {
99
+ "function": "take_first"
100
+ }
101
+ ]
102
+ },
103
+ {
104
+ "name": "flexible-extract",
105
+ "filter": [
106
+ {
107
+ "function": "regex",
108
+ "group_select": -1,
109
+ "regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
110
+ },
111
+ {
112
+ "function": "take_first"
113
+ }
114
+ ]
115
+ }
116
+ ],
117
+ "should_decontaminate": false,
118
+ "metadata": {
119
+ "version": 3.0,
120
+ "model": "student",
121
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
122
+ "num_concurrent": 48,
123
+ "tokenized_requests": false,
124
+ "max_retries": 3,
125
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
126
+ }
127
+ },
128
+ "ifeval": {
129
+ "task": "ifeval",
130
+ "dataset_path": "google/IFEval",
131
+ "test_split": "train",
132
+ "doc_to_text": "prompt",
133
+ "doc_to_target": 0,
134
+ "unsafe_code": false,
135
+ "process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
136
+ "description": "",
137
+ "target_delimiter": " ",
138
+ "fewshot_delimiter": "\n\n",
139
+ "fewshot_config": {
140
+ "sampler": "default",
141
+ "split": null,
142
+ "process_docs": null,
143
+ "fewshot_indices": null,
144
+ "samples": null,
145
+ "doc_to_text": "prompt",
146
+ "doc_to_choice": null,
147
+ "doc_to_target": 0,
148
+ "gen_prefix": null,
149
+ "fewshot_delimiter": "\n\n",
150
+ "target_delimiter": " "
151
+ },
152
+ "num_fewshot": 0,
153
+ "metric_list": [
154
+ {
155
+ "metric": "prompt_level_strict_acc",
156
+ "aggregation": "mean",
157
+ "higher_is_better": true
158
+ },
159
+ {
160
+ "metric": "inst_level_strict_acc",
161
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
162
+ "higher_is_better": true
163
+ },
164
+ {
165
+ "metric": "prompt_level_loose_acc",
166
+ "aggregation": "mean",
167
+ "higher_is_better": true
168
+ },
169
+ {
170
+ "metric": "inst_level_loose_acc",
171
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
172
+ "higher_is_better": true
173
+ }
174
+ ],
175
+ "output_type": "generate_until",
176
+ "generation_kwargs": {
177
+ "until": [],
178
+ "do_sample": false,
179
+ "temperature": 0.0,
180
+ "max_gen_toks": 1280
181
+ },
182
+ "repeats": 1,
183
+ "should_decontaminate": false,
184
+ "metadata": {
185
+ "version": 4.0,
186
+ "model": "student",
187
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
188
+ "num_concurrent": 48,
189
+ "tokenized_requests": false,
190
+ "max_retries": 3,
191
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
192
+ }
193
+ },
194
+ "minerva_math500": {
195
+ "task": "minerva_math500",
196
+ "dataset_path": "HuggingFaceH4/MATH-500",
197
+ "dataset_name": "default",
198
+ "test_split": "test",
199
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
200
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
201
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
202
+ "unsafe_code": false,
203
+ "process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
204
+ "description": "",
205
+ "target_delimiter": " ",
206
+ "fewshot_delimiter": "\n\n",
207
+ "fewshot_config": {
208
+ "sampler": "first_n",
209
+ "split": null,
210
+ "process_docs": "<function process_docs at 0x74c074c25b20>",
211
+ "fewshot_indices": null,
212
+ "samples": "<function list_fewshot_samples at 0x74c074c279c0>",
213
+ "doc_to_text": "<function doc_to_text at 0x74c0763f1800>",
214
+ "doc_to_choice": null,
215
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
216
+ "gen_prefix": null,
217
+ "fewshot_delimiter": "\n\n",
218
+ "target_delimiter": " "
219
+ },
220
+ "num_fewshot": 4,
221
+ "metric_list": [
222
+ {
223
+ "metric": "exact_match",
224
+ "aggregation": "mean",
225
+ "higher_is_better": true
226
+ },
227
+ {
228
+ "metric": "math_verify",
229
+ "aggregation": "mean",
230
+ "higher_is_better": true
231
+ }
232
+ ],
233
+ "output_type": "generate_until",
234
+ "generation_kwargs": {
235
+ "until": [
236
+ "Problem:"
237
+ ],
238
+ "do_sample": false,
239
+ "temperature": 0.0,
240
+ "max_gen_toks": 1280
241
+ },
242
+ "repeats": 1,
243
+ "should_decontaminate": false,
244
+ "metadata": {
245
+ "version": 3.0,
246
+ "model": "student",
247
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
248
+ "num_concurrent": 48,
249
+ "tokenized_requests": false,
250
+ "max_retries": 3,
251
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
252
+ }
253
+ }
254
+ },
255
+ "versions": {
256
+ "gsm8k_cot_zeroshot": 3.0,
257
+ "ifeval": 4.0,
258
+ "minerva_math500": 3.0
259
+ },
260
+ "n-shot": {
261
+ "gsm8k_cot_zeroshot": 0,
262
+ "ifeval": 0,
263
+ "minerva_math500": 4
264
+ },
265
+ "higher_is_better": {
266
+ "gsm8k_cot_zeroshot": {
267
+ "exact_match": true
268
+ },
269
+ "ifeval": {
270
+ "prompt_level_strict_acc": true,
271
+ "inst_level_strict_acc": true,
272
+ "prompt_level_loose_acc": true,
273
+ "inst_level_loose_acc": true
274
+ },
275
+ "minerva_math500": {
276
+ "exact_match": true,
277
+ "math_verify": true
278
+ }
279
+ },
280
+ "n-samples": {
281
+ "gsm8k_cot_zeroshot": {
282
+ "original": 1319,
283
+ "effective": 1319
284
+ },
285
+ "minerva_math500": {
286
+ "original": 500,
287
+ "effective": 500
288
+ },
289
+ "ifeval": {
290
+ "original": 541,
291
+ "effective": 541
292
+ }
293
+ },
294
+ "config": {
295
+ "model": "local-chat-completions",
296
+ "model_args": {
297
+ "model": "student",
298
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
299
+ "num_concurrent": 48,
300
+ "tokenized_requests": false,
301
+ "max_retries": 3
302
+ },
303
+ "batch_size": 1,
304
+ "batch_sizes": [],
305
+ "device": "cuda:0",
306
+ "use_cache": null,
307
+ "limit": null,
308
+ "bootstrap_iters": 100000,
309
+ "gen_kwargs": {
310
+ "max_gen_toks": 1280
311
+ },
312
+ "random_seed": 0,
313
+ "numpy_seed": 1234,
314
+ "torch_seed": 1234,
315
+ "fewshot_seed": 1234
316
+ },
317
+ "git_hash": "4cd474a",
318
+ "date": 1784521647.9592357,
319
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
320
+ "transformers_version": "4.57.6",
321
+ "lm_eval_version": "0.4.12",
322
+ "upper_git_hash": null,
323
+ "task_hashes": {
324
+ "gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
325
+ "minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
326
+ "ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
327
+ },
328
+ "model_source": "local-chat-completions",
329
+ "model_name": "student",
330
+ "model_name_sanitized": "student",
331
+ "system_instruction": null,
332
+ "system_instruction_sha": null,
333
+ "fewshot_as_multiturn": true,
334
+ "chat_template": "",
335
+ "chat_template_sha": null,
336
+ "total_evaluation_time_seconds": "563.8881401320104"
337
+ }
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/results_2026-07-19T21-40-17.349569.json ADDED
@@ -0,0 +1,225 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "humaneval": {
4
+ "name": "humaneval",
5
+ "alias": "humaneval",
6
+ "sample_len": 164,
7
+ "pass@1,create_test": 0.2804878048780488,
8
+ "pass@1_stderr,create_test": 0.0351870022880158
9
+ },
10
+ "mbpp": {
11
+ "name": "mbpp",
12
+ "alias": "mbpp",
13
+ "sample_len": 500,
14
+ "pass_at_1,none": 0.226,
15
+ "pass_at_1_stderr,none": 0.018722956449139915
16
+ }
17
+ },
18
+ "group_subtasks": {},
19
+ "configs": {
20
+ "humaneval": {
21
+ "task": "humaneval",
22
+ "dataset_path": "openai/openai_humaneval",
23
+ "test_split": "test",
24
+ "doc_to_text": "{{prompt}}",
25
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
26
+ "unsafe_code": true,
27
+ "description": "",
28
+ "target_delimiter": " ",
29
+ "fewshot_delimiter": "\n\n",
30
+ "fewshot_config": {
31
+ "sampler": "default",
32
+ "split": null,
33
+ "process_docs": null,
34
+ "fewshot_indices": null,
35
+ "samples": null,
36
+ "doc_to_text": "{{prompt}}",
37
+ "doc_to_choice": null,
38
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
39
+ "gen_prefix": null,
40
+ "fewshot_delimiter": "\n\n",
41
+ "target_delimiter": " "
42
+ },
43
+ "num_fewshot": 0,
44
+ "metric_list": [
45
+ {
46
+ "metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
47
+ "aggregation": "mean",
48
+ "higher_is_better": true,
49
+ "k": [
50
+ 1
51
+ ]
52
+ }
53
+ ],
54
+ "output_type": "generate_until",
55
+ "generation_kwargs": {
56
+ "until": [
57
+ "\nclass",
58
+ "\ndef",
59
+ "\n#",
60
+ "\nif",
61
+ "\nprint"
62
+ ],
63
+ "max_gen_toks": 1024,
64
+ "do_sample": false
65
+ },
66
+ "repeats": 1,
67
+ "filter_list": [
68
+ {
69
+ "name": "create_test",
70
+ "filter": [
71
+ {
72
+ "function": "custom",
73
+ "filter_fn": "<function build_predictions at 0x767d37f56ca0>"
74
+ }
75
+ ]
76
+ }
77
+ ],
78
+ "should_decontaminate": false,
79
+ "metadata": {
80
+ "version": 1.0,
81
+ "model": "student",
82
+ "base_url": "http://127.0.0.1:8422/v1/completions",
83
+ "tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0150",
84
+ "num_concurrent": 48,
85
+ "tokenized_requests": false,
86
+ "max_retries": 3,
87
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
88
+ }
89
+ },
90
+ "mbpp": {
91
+ "task": "mbpp",
92
+ "dataset_path": "google-research-datasets/mbpp",
93
+ "dataset_name": "full",
94
+ "test_split": "test",
95
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
96
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
97
+ "unsafe_code": true,
98
+ "description": "",
99
+ "target_delimiter": "",
100
+ "fewshot_delimiter": "\n\n",
101
+ "fewshot_config": {
102
+ "sampler": "first_n",
103
+ "split": null,
104
+ "process_docs": null,
105
+ "fewshot_indices": null,
106
+ "samples": "<function list_fewshot_samples at 0x767e00cef9c0>",
107
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
108
+ "doc_to_choice": null,
109
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
110
+ "gen_prefix": null,
111
+ "fewshot_delimiter": "\n\n",
112
+ "target_delimiter": ""
113
+ },
114
+ "num_fewshot": 3,
115
+ "metric_list": [
116
+ {
117
+ "metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
118
+ "aggregation": "mean",
119
+ "higher_is_better": true
120
+ }
121
+ ],
122
+ "output_type": "generate_until",
123
+ "generation_kwargs": {
124
+ "until": [
125
+ "[DONE]"
126
+ ],
127
+ "do_sample": false
128
+ },
129
+ "repeats": 1,
130
+ "should_decontaminate": false,
131
+ "metadata": {
132
+ "version": 1.0,
133
+ "model": "student",
134
+ "base_url": "http://127.0.0.1:8422/v1/completions",
135
+ "tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0150",
136
+ "num_concurrent": 48,
137
+ "tokenized_requests": false,
138
+ "max_retries": 3,
139
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
140
+ }
141
+ }
142
+ },
143
+ "versions": {
144
+ "humaneval": 1.0,
145
+ "mbpp": 1.0
146
+ },
147
+ "n-shot": {
148
+ "humaneval": 0,
149
+ "mbpp": 3
150
+ },
151
+ "higher_is_better": {
152
+ "humaneval": {
153
+ "pass_at_k": true,
154
+ "pass@1": true
155
+ },
156
+ "mbpp": {
157
+ "pass_at_1": true
158
+ }
159
+ },
160
+ "n-samples": {
161
+ "humaneval": {
162
+ "original": 164,
163
+ "effective": 164
164
+ },
165
+ "mbpp": {
166
+ "original": 500,
167
+ "effective": 500
168
+ }
169
+ },
170
+ "config": {
171
+ "model": "local-completions",
172
+ "model_args": {
173
+ "model": "student",
174
+ "base_url": "http://127.0.0.1:8422/v1/completions",
175
+ "tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0150",
176
+ "num_concurrent": 48,
177
+ "tokenized_requests": false,
178
+ "max_retries": 3
179
+ },
180
+ "batch_size": 1,
181
+ "batch_sizes": [],
182
+ "device": "cuda:0",
183
+ "use_cache": null,
184
+ "limit": null,
185
+ "bootstrap_iters": 100000,
186
+ "gen_kwargs": {},
187
+ "random_seed": 0,
188
+ "numpy_seed": 1234,
189
+ "torch_seed": 1234,
190
+ "fewshot_seed": 1234
191
+ },
192
+ "git_hash": "4cd474a",
193
+ "date": 1784522214.7088265,
194
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
195
+ "transformers_version": "4.57.6",
196
+ "lm_eval_version": "0.4.12",
197
+ "upper_git_hash": null,
198
+ "tokenizer_pad_token": [
199
+ "<pad>",
200
+ "50280"
201
+ ],
202
+ "tokenizer_eos_token": [
203
+ "|||IP_ADDRESS|||",
204
+ "50279"
205
+ ],
206
+ "tokenizer_bos_token": [
207
+ "|||IP_ADDRESS|||",
208
+ "50279"
209
+ ],
210
+ "eot_token_id": 50279,
211
+ "max_length": 2047,
212
+ "task_hashes": {
213
+ "humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
214
+ "mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
215
+ },
216
+ "model_source": "local-completions",
217
+ "model_name": "student",
218
+ "model_name_sanitized": "student",
219
+ "system_instruction": null,
220
+ "system_instruction_sha": null,
221
+ "fewshot_as_multiturn": null,
222
+ "chat_template": null,
223
+ "chat_template_sha": null,
224
+ "total_evaluation_time_seconds": "208.97489892301382"
225
+ }
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_gsm8k_cot_zeroshot_2026-07-19T21-36-45.489848.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_humaneval_2026-07-19T21-40-17.349569.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_ifeval_2026-07-19T21-36-45.489848.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_mbpp_2026-07-19T21-40-17.349569.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224_long500_step150/student/samples_minerva_math500_2026-07-19T21-36-45.489848.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/results_2026-07-19T21-49-11.345128.json ADDED
@@ -0,0 +1,337 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "gsm8k_cot_zeroshot": {
4
+ "name": "gsm8k_cot_zeroshot",
5
+ "alias": "gsm8k_cot_zeroshot",
6
+ "sample_len": 1319,
7
+ "exact_match,strict-match": 0.0,
8
+ "exact_match_stderr,strict-match": 0.0,
9
+ "exact_match,flexible-extract": 0.558756633813495,
10
+ "exact_match_stderr,flexible-extract": 0.01367705947859264
11
+ },
12
+ "minerva_math500": {
13
+ "name": "minerva_math500",
14
+ "alias": "minerva_math500",
15
+ "sample_len": 500,
16
+ "exact_match,none": 0.138,
17
+ "exact_match_stderr,none": 0.015439843831953423,
18
+ "math_verify,none": 0.208,
19
+ "math_verify_stderr,none": 0.01816954222122996
20
+ },
21
+ "ifeval": {
22
+ "name": "ifeval",
23
+ "alias": "ifeval",
24
+ "sample_len": 541,
25
+ "prompt_level_strict_acc,none": 0.5730129390018485,
26
+ "prompt_level_strict_acc_stderr,none": 0.021285933050061243,
27
+ "inst_level_strict_acc,none": 0.6870503597122302,
28
+ "inst_level_strict_acc_stderr,none": "N/A",
29
+ "prompt_level_loose_acc,none": 0.5933456561922366,
30
+ "prompt_level_loose_acc_stderr,none": 0.021138283177336344,
31
+ "inst_level_loose_acc,none": 0.7086330935251799,
32
+ "inst_level_loose_acc_stderr,none": "N/A"
33
+ }
34
+ },
35
+ "group_subtasks": {},
36
+ "configs": {
37
+ "gsm8k_cot_zeroshot": {
38
+ "task": "gsm8k_cot_zeroshot",
39
+ "dataset_path": "openai/gsm8k",
40
+ "dataset_name": "main",
41
+ "training_split": "train",
42
+ "test_split": "test",
43
+ "fewshot_split": "train",
44
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
45
+ "doc_to_target": "{{answer}}",
46
+ "unsafe_code": false,
47
+ "description": "",
48
+ "target_delimiter": " ",
49
+ "fewshot_delimiter": "\n\n",
50
+ "fewshot_config": {
51
+ "sampler": "default",
52
+ "split": "train",
53
+ "process_docs": null,
54
+ "fewshot_indices": null,
55
+ "samples": null,
56
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
57
+ "doc_to_choice": null,
58
+ "doc_to_target": "{{answer}}",
59
+ "gen_prefix": null,
60
+ "fewshot_delimiter": "\n\n",
61
+ "target_delimiter": " "
62
+ },
63
+ "num_fewshot": 0,
64
+ "metric_list": [
65
+ {
66
+ "metric": "exact_match",
67
+ "aggregation": "mean",
68
+ "higher_is_better": true,
69
+ "ignore_case": true,
70
+ "ignore_punctuation": false,
71
+ "regexes_to_ignore": [
72
+ ",",
73
+ "\\$",
74
+ "(?s).*#### ",
75
+ "\\.$"
76
+ ]
77
+ }
78
+ ],
79
+ "output_type": "generate_until",
80
+ "generation_kwargs": {
81
+ "until": [
82
+ "Q:",
83
+ "</s>",
84
+ "<|im_end|>"
85
+ ],
86
+ "do_sample": false,
87
+ "max_gen_toks": 1280
88
+ },
89
+ "repeats": 1,
90
+ "filter_list": [
91
+ {
92
+ "name": "strict-match",
93
+ "filter": [
94
+ {
95
+ "function": "regex",
96
+ "regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
97
+ },
98
+ {
99
+ "function": "take_first"
100
+ }
101
+ ]
102
+ },
103
+ {
104
+ "name": "flexible-extract",
105
+ "filter": [
106
+ {
107
+ "function": "regex",
108
+ "group_select": -1,
109
+ "regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
110
+ },
111
+ {
112
+ "function": "take_first"
113
+ }
114
+ ]
115
+ }
116
+ ],
117
+ "should_decontaminate": false,
118
+ "metadata": {
119
+ "version": 3.0,
120
+ "model": "student",
121
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
122
+ "num_concurrent": 48,
123
+ "tokenized_requests": false,
124
+ "max_retries": 3,
125
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
126
+ }
127
+ },
128
+ "ifeval": {
129
+ "task": "ifeval",
130
+ "dataset_path": "google/IFEval",
131
+ "test_split": "train",
132
+ "doc_to_text": "prompt",
133
+ "doc_to_target": 0,
134
+ "unsafe_code": false,
135
+ "process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
136
+ "description": "",
137
+ "target_delimiter": " ",
138
+ "fewshot_delimiter": "\n\n",
139
+ "fewshot_config": {
140
+ "sampler": "default",
141
+ "split": null,
142
+ "process_docs": null,
143
+ "fewshot_indices": null,
144
+ "samples": null,
145
+ "doc_to_text": "prompt",
146
+ "doc_to_choice": null,
147
+ "doc_to_target": 0,
148
+ "gen_prefix": null,
149
+ "fewshot_delimiter": "\n\n",
150
+ "target_delimiter": " "
151
+ },
152
+ "num_fewshot": 0,
153
+ "metric_list": [
154
+ {
155
+ "metric": "prompt_level_strict_acc",
156
+ "aggregation": "mean",
157
+ "higher_is_better": true
158
+ },
159
+ {
160
+ "metric": "inst_level_strict_acc",
161
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
162
+ "higher_is_better": true
163
+ },
164
+ {
165
+ "metric": "prompt_level_loose_acc",
166
+ "aggregation": "mean",
167
+ "higher_is_better": true
168
+ },
169
+ {
170
+ "metric": "inst_level_loose_acc",
171
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
172
+ "higher_is_better": true
173
+ }
174
+ ],
175
+ "output_type": "generate_until",
176
+ "generation_kwargs": {
177
+ "until": [],
178
+ "do_sample": false,
179
+ "temperature": 0.0,
180
+ "max_gen_toks": 1280
181
+ },
182
+ "repeats": 1,
183
+ "should_decontaminate": false,
184
+ "metadata": {
185
+ "version": 4.0,
186
+ "model": "student",
187
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
188
+ "num_concurrent": 48,
189
+ "tokenized_requests": false,
190
+ "max_retries": 3,
191
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
192
+ }
193
+ },
194
+ "minerva_math500": {
195
+ "task": "minerva_math500",
196
+ "dataset_path": "HuggingFaceH4/MATH-500",
197
+ "dataset_name": "default",
198
+ "test_split": "test",
199
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
200
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
201
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
202
+ "unsafe_code": false,
203
+ "process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
204
+ "description": "",
205
+ "target_delimiter": " ",
206
+ "fewshot_delimiter": "\n\n",
207
+ "fewshot_config": {
208
+ "sampler": "first_n",
209
+ "split": null,
210
+ "process_docs": "<function process_docs at 0x750976e35b20>",
211
+ "fewshot_indices": null,
212
+ "samples": "<function list_fewshot_samples at 0x750976e379c0>",
213
+ "doc_to_text": "<function doc_to_text at 0x75097c5fd800>",
214
+ "doc_to_choice": null,
215
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
216
+ "gen_prefix": null,
217
+ "fewshot_delimiter": "\n\n",
218
+ "target_delimiter": " "
219
+ },
220
+ "num_fewshot": 4,
221
+ "metric_list": [
222
+ {
223
+ "metric": "exact_match",
224
+ "aggregation": "mean",
225
+ "higher_is_better": true
226
+ },
227
+ {
228
+ "metric": "math_verify",
229
+ "aggregation": "mean",
230
+ "higher_is_better": true
231
+ }
232
+ ],
233
+ "output_type": "generate_until",
234
+ "generation_kwargs": {
235
+ "until": [
236
+ "Problem:"
237
+ ],
238
+ "do_sample": false,
239
+ "temperature": 0.0,
240
+ "max_gen_toks": 1280
241
+ },
242
+ "repeats": 1,
243
+ "should_decontaminate": false,
244
+ "metadata": {
245
+ "version": 3.0,
246
+ "model": "student",
247
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
248
+ "num_concurrent": 48,
249
+ "tokenized_requests": false,
250
+ "max_retries": 3,
251
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
252
+ }
253
+ }
254
+ },
255
+ "versions": {
256
+ "gsm8k_cot_zeroshot": 3.0,
257
+ "ifeval": 4.0,
258
+ "minerva_math500": 3.0
259
+ },
260
+ "n-shot": {
261
+ "gsm8k_cot_zeroshot": 0,
262
+ "ifeval": 0,
263
+ "minerva_math500": 4
264
+ },
265
+ "higher_is_better": {
266
+ "gsm8k_cot_zeroshot": {
267
+ "exact_match": true
268
+ },
269
+ "ifeval": {
270
+ "prompt_level_strict_acc": true,
271
+ "inst_level_strict_acc": true,
272
+ "prompt_level_loose_acc": true,
273
+ "inst_level_loose_acc": true
274
+ },
275
+ "minerva_math500": {
276
+ "exact_match": true,
277
+ "math_verify": true
278
+ }
279
+ },
280
+ "n-samples": {
281
+ "gsm8k_cot_zeroshot": {
282
+ "original": 1319,
283
+ "effective": 1319
284
+ },
285
+ "minerva_math500": {
286
+ "original": 500,
287
+ "effective": 500
288
+ },
289
+ "ifeval": {
290
+ "original": 541,
291
+ "effective": 541
292
+ }
293
+ },
294
+ "config": {
295
+ "model": "local-chat-completions",
296
+ "model_args": {
297
+ "model": "student",
298
+ "base_url": "http://127.0.0.1:8422/v1/chat/completions",
299
+ "num_concurrent": 48,
300
+ "tokenized_requests": false,
301
+ "max_retries": 3
302
+ },
303
+ "batch_size": 1,
304
+ "batch_sizes": [],
305
+ "device": "cuda:0",
306
+ "use_cache": null,
307
+ "limit": null,
308
+ "bootstrap_iters": 100000,
309
+ "gen_kwargs": {
310
+ "max_gen_toks": 1280
311
+ },
312
+ "random_seed": 0,
313
+ "numpy_seed": 1234,
314
+ "torch_seed": 1234,
315
+ "fewshot_seed": 1234
316
+ },
317
+ "git_hash": "4cd474a",
318
+ "date": 1784522456.8447104,
319
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
320
+ "transformers_version": "4.57.6",
321
+ "lm_eval_version": "0.4.12",
322
+ "upper_git_hash": null,
323
+ "task_hashes": {
324
+ "gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
325
+ "minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
326
+ "ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
327
+ },
328
+ "model_source": "local-chat-completions",
329
+ "model_name": "student",
330
+ "model_name_sanitized": "student",
331
+ "system_instruction": null,
332
+ "system_instruction_sha": null,
333
+ "fewshot_as_multiturn": true,
334
+ "chat_template": "",
335
+ "chat_template_sha": null,
336
+ "total_evaluation_time_seconds": "500.82202401198447"
337
+ }
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/results_2026-07-19T21-52-17.977663.json ADDED
@@ -0,0 +1,225 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "humaneval": {
4
+ "name": "humaneval",
5
+ "alias": "humaneval",
6
+ "sample_len": 164,
7
+ "pass@1,create_test": 0.3170731707317073,
8
+ "pass@1_stderr,create_test": 0.03644794381282879
9
+ },
10
+ "mbpp": {
11
+ "name": "mbpp",
12
+ "alias": "mbpp",
13
+ "sample_len": 500,
14
+ "pass_at_1,none": 0.234,
15
+ "pass_at_1_stderr,none": 0.01895274156489368
16
+ }
17
+ },
18
+ "group_subtasks": {},
19
+ "configs": {
20
+ "humaneval": {
21
+ "task": "humaneval",
22
+ "dataset_path": "openai/openai_humaneval",
23
+ "test_split": "test",
24
+ "doc_to_text": "{{prompt}}",
25
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
26
+ "unsafe_code": true,
27
+ "description": "",
28
+ "target_delimiter": " ",
29
+ "fewshot_delimiter": "\n\n",
30
+ "fewshot_config": {
31
+ "sampler": "default",
32
+ "split": null,
33
+ "process_docs": null,
34
+ "fewshot_indices": null,
35
+ "samples": null,
36
+ "doc_to_text": "{{prompt}}",
37
+ "doc_to_choice": null,
38
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
39
+ "gen_prefix": null,
40
+ "fewshot_delimiter": "\n\n",
41
+ "target_delimiter": " "
42
+ },
43
+ "num_fewshot": 0,
44
+ "metric_list": [
45
+ {
46
+ "metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
47
+ "aggregation": "mean",
48
+ "higher_is_better": true,
49
+ "k": [
50
+ 1
51
+ ]
52
+ }
53
+ ],
54
+ "output_type": "generate_until",
55
+ "generation_kwargs": {
56
+ "until": [
57
+ "\nclass",
58
+ "\ndef",
59
+ "\n#",
60
+ "\nif",
61
+ "\nprint"
62
+ ],
63
+ "max_gen_toks": 1024,
64
+ "do_sample": false
65
+ },
66
+ "repeats": 1,
67
+ "filter_list": [
68
+ {
69
+ "name": "create_test",
70
+ "filter": [
71
+ {
72
+ "function": "custom",
73
+ "filter_fn": "<function build_predictions at 0x7de0f336eca0>"
74
+ }
75
+ ]
76
+ }
77
+ ],
78
+ "should_decontaminate": false,
79
+ "metadata": {
80
+ "version": 1.0,
81
+ "model": "student",
82
+ "base_url": "http://127.0.0.1:8422/v1/completions",
83
+ "tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0500",
84
+ "num_concurrent": 48,
85
+ "tokenized_requests": false,
86
+ "max_retries": 3,
87
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
88
+ }
89
+ },
90
+ "mbpp": {
91
+ "task": "mbpp",
92
+ "dataset_path": "google-research-datasets/mbpp",
93
+ "dataset_name": "full",
94
+ "test_split": "test",
95
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
96
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
97
+ "unsafe_code": true,
98
+ "description": "",
99
+ "target_delimiter": "",
100
+ "fewshot_delimiter": "\n\n",
101
+ "fewshot_config": {
102
+ "sampler": "first_n",
103
+ "split": null,
104
+ "process_docs": null,
105
+ "fewshot_indices": null,
106
+ "samples": "<function list_fewshot_samples at 0x7de1bc5079c0>",
107
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
108
+ "doc_to_choice": null,
109
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
110
+ "gen_prefix": null,
111
+ "fewshot_delimiter": "\n\n",
112
+ "target_delimiter": ""
113
+ },
114
+ "num_fewshot": 3,
115
+ "metric_list": [
116
+ {
117
+ "metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
118
+ "aggregation": "mean",
119
+ "higher_is_better": true
120
+ }
121
+ ],
122
+ "output_type": "generate_until",
123
+ "generation_kwargs": {
124
+ "until": [
125
+ "[DONE]"
126
+ ],
127
+ "do_sample": false
128
+ },
129
+ "repeats": 1,
130
+ "should_decontaminate": false,
131
+ "metadata": {
132
+ "version": 1.0,
133
+ "model": "student",
134
+ "base_url": "http://127.0.0.1:8422/v1/completions",
135
+ "tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0500",
136
+ "num_concurrent": 48,
137
+ "tokenized_requests": false,
138
+ "max_retries": 3,
139
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
140
+ }
141
+ }
142
+ },
143
+ "versions": {
144
+ "humaneval": 1.0,
145
+ "mbpp": 1.0
146
+ },
147
+ "n-shot": {
148
+ "humaneval": 0,
149
+ "mbpp": 3
150
+ },
151
+ "higher_is_better": {
152
+ "humaneval": {
153
+ "pass_at_k": true,
154
+ "pass@1": true
155
+ },
156
+ "mbpp": {
157
+ "pass_at_1": true
158
+ }
159
+ },
160
+ "n-samples": {
161
+ "humaneval": {
162
+ "original": 164,
163
+ "effective": 164
164
+ },
165
+ "mbpp": {
166
+ "original": 500,
167
+ "effective": 500
168
+ }
169
+ },
170
+ "config": {
171
+ "model": "local-completions",
172
+ "model_args": {
173
+ "model": "student",
174
+ "base_url": "http://127.0.0.1:8422/v1/completions",
175
+ "tokenizer": "outputs/healed/grid_general_fairness/glean_keep50_s1224_long500/step0500",
176
+ "num_concurrent": 48,
177
+ "tokenized_requests": false,
178
+ "max_retries": 3
179
+ },
180
+ "batch_size": 1,
181
+ "batch_sizes": [],
182
+ "device": "cuda:0",
183
+ "use_cache": null,
184
+ "limit": null,
185
+ "bootstrap_iters": 100000,
186
+ "gen_kwargs": {},
187
+ "random_seed": 0,
188
+ "numpy_seed": 1234,
189
+ "torch_seed": 1234,
190
+ "fewshot_seed": 1234
191
+ },
192
+ "git_hash": "4cd474a",
193
+ "date": 1784522960.7513723,
194
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
195
+ "transformers_version": "4.57.6",
196
+ "lm_eval_version": "0.4.12",
197
+ "upper_git_hash": null,
198
+ "tokenizer_pad_token": [
199
+ "<pad>",
200
+ "50280"
201
+ ],
202
+ "tokenizer_eos_token": [
203
+ "|||IP_ADDRESS|||",
204
+ "50279"
205
+ ],
206
+ "tokenizer_bos_token": [
207
+ "|||IP_ADDRESS|||",
208
+ "50279"
209
+ ],
210
+ "eot_token_id": 50279,
211
+ "max_length": 2047,
212
+ "task_hashes": {
213
+ "humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
214
+ "mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
215
+ },
216
+ "model_source": "local-completions",
217
+ "model_name": "student",
218
+ "model_name_sanitized": "student",
219
+ "system_instruction": null,
220
+ "system_instruction_sha": null,
221
+ "fewshot_as_multiturn": null,
222
+ "chat_template": null,
223
+ "chat_template_sha": null,
224
+ "total_evaluation_time_seconds": "183.55893040599767"
225
+ }
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_gsm8k_cot_zeroshot_2026-07-19T21-49-11.345128.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_humaneval_2026-07-19T21-52-17.977663.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_ifeval_2026-07-19T21-49-11.345128.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_mbpp_2026-07-19T21-52-17.977663.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224_long500_step500/student/samples_minerva_math500_2026-07-19T21-49-11.345128.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
evals/general_suite/healed/glean_keep50_s1224_step100/student/results_2026-07-18T10-26-00.525964.json ADDED
@@ -0,0 +1,337 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "gsm8k_cot_zeroshot": {
4
+ "name": "gsm8k_cot_zeroshot",
5
+ "alias": "gsm8k_cot_zeroshot",
6
+ "sample_len": 1319,
7
+ "exact_match,strict-match": 0.0,
8
+ "exact_match_stderr,strict-match": 0.0,
9
+ "exact_match,flexible-extract": 0.535253980288097,
10
+ "exact_match_stderr,flexible-extract": 0.013738207990177317
11
+ },
12
+ "minerva_math500": {
13
+ "name": "minerva_math500",
14
+ "alias": "minerva_math500",
15
+ "sample_len": 500,
16
+ "exact_match,none": 0.124,
17
+ "exact_match_stderr,none": 0.014754096608517476,
18
+ "math_verify,none": 0.186,
19
+ "math_verify_stderr,none": 0.01741880678058399
20
+ },
21
+ "ifeval": {
22
+ "name": "ifeval",
23
+ "alias": "ifeval",
24
+ "sample_len": 541,
25
+ "prompt_level_strict_acc,none": 0.5415896487985212,
26
+ "prompt_level_strict_acc_stderr,none": 0.021442010560476468,
27
+ "inst_level_strict_acc,none": 0.6642685851318945,
28
+ "inst_level_strict_acc_stderr,none": "N/A",
29
+ "prompt_level_loose_acc,none": 0.5748613678373382,
30
+ "prompt_level_loose_acc_stderr,none": 0.021274039805355655,
31
+ "inst_level_loose_acc,none": 0.6954436450839329,
32
+ "inst_level_loose_acc_stderr,none": "N/A"
33
+ }
34
+ },
35
+ "group_subtasks": {},
36
+ "configs": {
37
+ "gsm8k_cot_zeroshot": {
38
+ "task": "gsm8k_cot_zeroshot",
39
+ "dataset_path": "openai/gsm8k",
40
+ "dataset_name": "main",
41
+ "training_split": "train",
42
+ "test_split": "test",
43
+ "fewshot_split": "train",
44
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
45
+ "doc_to_target": "{{answer}}",
46
+ "unsafe_code": false,
47
+ "description": "",
48
+ "target_delimiter": " ",
49
+ "fewshot_delimiter": "\n\n",
50
+ "fewshot_config": {
51
+ "sampler": "default",
52
+ "split": "train",
53
+ "process_docs": null,
54
+ "fewshot_indices": null,
55
+ "samples": null,
56
+ "doc_to_text": "Q: {{question}}\nA: Let's think step by step.",
57
+ "doc_to_choice": null,
58
+ "doc_to_target": "{{answer}}",
59
+ "gen_prefix": null,
60
+ "fewshot_delimiter": "\n\n",
61
+ "target_delimiter": " "
62
+ },
63
+ "num_fewshot": 0,
64
+ "metric_list": [
65
+ {
66
+ "metric": "exact_match",
67
+ "aggregation": "mean",
68
+ "higher_is_better": true,
69
+ "ignore_case": true,
70
+ "ignore_punctuation": false,
71
+ "regexes_to_ignore": [
72
+ ",",
73
+ "\\$",
74
+ "(?s).*#### ",
75
+ "\\.$"
76
+ ]
77
+ }
78
+ ],
79
+ "output_type": "generate_until",
80
+ "generation_kwargs": {
81
+ "until": [
82
+ "Q:",
83
+ "</s>",
84
+ "<|im_end|>"
85
+ ],
86
+ "do_sample": false,
87
+ "max_gen_toks": 1280
88
+ },
89
+ "repeats": 1,
90
+ "filter_list": [
91
+ {
92
+ "name": "strict-match",
93
+ "filter": [
94
+ {
95
+ "function": "regex",
96
+ "regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
97
+ },
98
+ {
99
+ "function": "take_first"
100
+ }
101
+ ]
102
+ },
103
+ {
104
+ "name": "flexible-extract",
105
+ "filter": [
106
+ {
107
+ "function": "regex",
108
+ "group_select": -1,
109
+ "regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
110
+ },
111
+ {
112
+ "function": "take_first"
113
+ }
114
+ ]
115
+ }
116
+ ],
117
+ "should_decontaminate": false,
118
+ "metadata": {
119
+ "version": 3.0,
120
+ "model": "student",
121
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
122
+ "num_concurrent": 48,
123
+ "tokenized_requests": false,
124
+ "max_retries": 3,
125
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml"
126
+ }
127
+ },
128
+ "ifeval": {
129
+ "task": "ifeval",
130
+ "dataset_path": "google/IFEval",
131
+ "test_split": "train",
132
+ "doc_to_text": "prompt",
133
+ "doc_to_target": 0,
134
+ "unsafe_code": false,
135
+ "process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
136
+ "description": "",
137
+ "target_delimiter": " ",
138
+ "fewshot_delimiter": "\n\n",
139
+ "fewshot_config": {
140
+ "sampler": "default",
141
+ "split": null,
142
+ "process_docs": null,
143
+ "fewshot_indices": null,
144
+ "samples": null,
145
+ "doc_to_text": "prompt",
146
+ "doc_to_choice": null,
147
+ "doc_to_target": 0,
148
+ "gen_prefix": null,
149
+ "fewshot_delimiter": "\n\n",
150
+ "target_delimiter": " "
151
+ },
152
+ "num_fewshot": 0,
153
+ "metric_list": [
154
+ {
155
+ "metric": "prompt_level_strict_acc",
156
+ "aggregation": "mean",
157
+ "higher_is_better": true
158
+ },
159
+ {
160
+ "metric": "inst_level_strict_acc",
161
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
162
+ "higher_is_better": true
163
+ },
164
+ {
165
+ "metric": "prompt_level_loose_acc",
166
+ "aggregation": "mean",
167
+ "higher_is_better": true
168
+ },
169
+ {
170
+ "metric": "inst_level_loose_acc",
171
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
172
+ "higher_is_better": true
173
+ }
174
+ ],
175
+ "output_type": "generate_until",
176
+ "generation_kwargs": {
177
+ "until": [],
178
+ "do_sample": false,
179
+ "temperature": 0.0,
180
+ "max_gen_toks": 1280
181
+ },
182
+ "repeats": 1,
183
+ "should_decontaminate": false,
184
+ "metadata": {
185
+ "version": 4.0,
186
+ "model": "student",
187
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
188
+ "num_concurrent": 48,
189
+ "tokenized_requests": false,
190
+ "max_retries": 3,
191
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/ifeval/ifeval.yaml"
192
+ }
193
+ },
194
+ "minerva_math500": {
195
+ "task": "minerva_math500",
196
+ "dataset_path": "HuggingFaceH4/MATH-500",
197
+ "dataset_name": "default",
198
+ "test_split": "test",
199
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
200
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
201
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
202
+ "unsafe_code": false,
203
+ "process_results": "def process_results(doc: dict, results: list[str]) -> dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n _mvres = verify(\n gold=parse(doc[\"solution\"]),\n target=parse(candidates),\n )\n mathval = 1 if _mvres else 0\n\n res = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return res\n",
204
+ "description": "",
205
+ "target_delimiter": " ",
206
+ "fewshot_delimiter": "\n\n",
207
+ "fewshot_config": {
208
+ "sampler": "first_n",
209
+ "split": null,
210
+ "process_docs": "<function process_docs at 0x7f32b8039b20>",
211
+ "fewshot_indices": null,
212
+ "samples": "<function list_fewshot_samples at 0x7f32b803b9c0>",
213
+ "doc_to_text": "<function doc_to_text at 0x7f32b9601800>",
214
+ "doc_to_choice": null,
215
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
216
+ "gen_prefix": null,
217
+ "fewshot_delimiter": "\n\n",
218
+ "target_delimiter": " "
219
+ },
220
+ "num_fewshot": 4,
221
+ "metric_list": [
222
+ {
223
+ "metric": "exact_match",
224
+ "aggregation": "mean",
225
+ "higher_is_better": true
226
+ },
227
+ {
228
+ "metric": "math_verify",
229
+ "aggregation": "mean",
230
+ "higher_is_better": true
231
+ }
232
+ ],
233
+ "output_type": "generate_until",
234
+ "generation_kwargs": {
235
+ "until": [
236
+ "Problem:"
237
+ ],
238
+ "do_sample": false,
239
+ "temperature": 0.0,
240
+ "max_gen_toks": 1280
241
+ },
242
+ "repeats": 1,
243
+ "should_decontaminate": false,
244
+ "metadata": {
245
+ "version": 3.0,
246
+ "model": "student",
247
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
248
+ "num_concurrent": 48,
249
+ "tokenized_requests": false,
250
+ "max_retries": 3,
251
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/minerva_math500.yaml"
252
+ }
253
+ }
254
+ },
255
+ "versions": {
256
+ "gsm8k_cot_zeroshot": 3.0,
257
+ "ifeval": 4.0,
258
+ "minerva_math500": 3.0
259
+ },
260
+ "n-shot": {
261
+ "gsm8k_cot_zeroshot": 0,
262
+ "ifeval": 0,
263
+ "minerva_math500": 4
264
+ },
265
+ "higher_is_better": {
266
+ "gsm8k_cot_zeroshot": {
267
+ "exact_match": true
268
+ },
269
+ "ifeval": {
270
+ "prompt_level_strict_acc": true,
271
+ "inst_level_strict_acc": true,
272
+ "prompt_level_loose_acc": true,
273
+ "inst_level_loose_acc": true
274
+ },
275
+ "minerva_math500": {
276
+ "exact_match": true,
277
+ "math_verify": true
278
+ }
279
+ },
280
+ "n-samples": {
281
+ "gsm8k_cot_zeroshot": {
282
+ "original": 1319,
283
+ "effective": 1319
284
+ },
285
+ "minerva_math500": {
286
+ "original": 500,
287
+ "effective": 500
288
+ },
289
+ "ifeval": {
290
+ "original": 541,
291
+ "effective": 541
292
+ }
293
+ },
294
+ "config": {
295
+ "model": "local-chat-completions",
296
+ "model_args": {
297
+ "model": "student",
298
+ "base_url": "http://127.0.0.1:8420/v1/chat/completions",
299
+ "num_concurrent": 48,
300
+ "tokenized_requests": false,
301
+ "max_retries": 3
302
+ },
303
+ "batch_size": 1,
304
+ "batch_sizes": [],
305
+ "device": "cuda:0",
306
+ "use_cache": null,
307
+ "limit": null,
308
+ "bootstrap_iters": 100000,
309
+ "gen_kwargs": {
310
+ "max_gen_toks": 1280
311
+ },
312
+ "random_seed": 0,
313
+ "numpy_seed": 1234,
314
+ "torch_seed": 1234,
315
+ "fewshot_seed": 1234
316
+ },
317
+ "git_hash": "4cd474a",
318
+ "date": 1784394960.8878376,
319
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
320
+ "transformers_version": "4.57.6",
321
+ "lm_eval_version": "0.4.12",
322
+ "upper_git_hash": null,
323
+ "task_hashes": {
324
+ "gsm8k_cot_zeroshot": "77ad80425a4cb85562b27f164a9c92977f775bb66a0ec4d374e4ee7b4656a7f5",
325
+ "minerva_math500": "5784cb060f783763ec8f8e234eef61589d348055c48e8adc0073445383258f85",
326
+ "ifeval": "bdcf10fc42fb3b39fd60af9d347a998b5bcb0a6782900c760b7dcc67ef0109d7"
327
+ },
328
+ "model_source": "local-chat-completions",
329
+ "model_name": "student",
330
+ "model_name_sanitized": "student",
331
+ "system_instruction": null,
332
+ "system_instruction_sha": null,
333
+ "fewshot_as_multiturn": true,
334
+ "chat_template": "",
335
+ "chat_template_sha": null,
336
+ "total_evaluation_time_seconds": "605.9618037229993"
337
+ }
evals/general_suite/healed/glean_keep50_s1224_step100/student/results_2026-07-18T10-29-19.004514.json ADDED
@@ -0,0 +1,225 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "humaneval": {
4
+ "name": "humaneval",
5
+ "alias": "humaneval",
6
+ "sample_len": 164,
7
+ "pass@1,create_test": 0.29878048780487804,
8
+ "pass@1_stderr,create_test": 0.035851663369096606
9
+ },
10
+ "mbpp": {
11
+ "name": "mbpp",
12
+ "alias": "mbpp",
13
+ "sample_len": 500,
14
+ "pass_at_1,none": 0.218,
15
+ "pass_at_1_stderr,none": 0.018483378223178866
16
+ }
17
+ },
18
+ "group_subtasks": {},
19
+ "configs": {
20
+ "humaneval": {
21
+ "task": "humaneval",
22
+ "dataset_path": "openai/openai_humaneval",
23
+ "test_split": "test",
24
+ "doc_to_text": "{{prompt}}",
25
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
26
+ "unsafe_code": true,
27
+ "description": "",
28
+ "target_delimiter": " ",
29
+ "fewshot_delimiter": "\n\n",
30
+ "fewshot_config": {
31
+ "sampler": "default",
32
+ "split": null,
33
+ "process_docs": null,
34
+ "fewshot_indices": null,
35
+ "samples": null,
36
+ "doc_to_text": "{{prompt}}",
37
+ "doc_to_choice": null,
38
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
39
+ "gen_prefix": null,
40
+ "fewshot_delimiter": "\n\n",
41
+ "target_delimiter": " "
42
+ },
43
+ "num_fewshot": 0,
44
+ "metric_list": [
45
+ {
46
+ "metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
47
+ "aggregation": "mean",
48
+ "higher_is_better": true,
49
+ "k": [
50
+ 1
51
+ ]
52
+ }
53
+ ],
54
+ "output_type": "generate_until",
55
+ "generation_kwargs": {
56
+ "until": [
57
+ "\nclass",
58
+ "\ndef",
59
+ "\n#",
60
+ "\nif",
61
+ "\nprint"
62
+ ],
63
+ "max_gen_toks": 1024,
64
+ "do_sample": false
65
+ },
66
+ "repeats": 1,
67
+ "filter_list": [
68
+ {
69
+ "name": "create_test",
70
+ "filter": [
71
+ {
72
+ "function": "custom",
73
+ "filter_fn": "<function build_predictions at 0x75dd03f1eca0>"
74
+ }
75
+ ]
76
+ }
77
+ ],
78
+ "should_decontaminate": false,
79
+ "metadata": {
80
+ "version": 1.0,
81
+ "model": "student",
82
+ "base_url": "http://127.0.0.1:8420/v1/completions",
83
+ "tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0100",
84
+ "num_concurrent": 48,
85
+ "tokenized_requests": false,
86
+ "max_retries": 3,
87
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/humaneval/humaneval.yaml"
88
+ }
89
+ },
90
+ "mbpp": {
91
+ "task": "mbpp",
92
+ "dataset_path": "google-research-datasets/mbpp",
93
+ "dataset_name": "full",
94
+ "test_split": "test",
95
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
96
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
97
+ "unsafe_code": true,
98
+ "description": "",
99
+ "target_delimiter": "",
100
+ "fewshot_delimiter": "\n\n",
101
+ "fewshot_config": {
102
+ "sampler": "first_n",
103
+ "split": null,
104
+ "process_docs": null,
105
+ "fewshot_indices": null,
106
+ "samples": "<function list_fewshot_samples at 0x75ddcceff9c0>",
107
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
108
+ "doc_to_choice": null,
109
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
110
+ "gen_prefix": null,
111
+ "fewshot_delimiter": "\n\n",
112
+ "target_delimiter": ""
113
+ },
114
+ "num_fewshot": 3,
115
+ "metric_list": [
116
+ {
117
+ "metric": "def pass_at_1(\n references: Union[str, list[str]], predictions: Union[str, list[list[str]]]\n) -> float:\n if isinstance(references, str):\n references = [references]\n if isinstance(predictions[0], str):\n predictions = [[p] for p in predictions]\n return pass_at_k.compute(\n references=references,\n predictions=predictions,\n k=[1],\n )[0][\"pass@1\"]\n",
118
+ "aggregation": "mean",
119
+ "higher_is_better": true
120
+ }
121
+ ],
122
+ "output_type": "generate_until",
123
+ "generation_kwargs": {
124
+ "until": [
125
+ "[DONE]"
126
+ ],
127
+ "do_sample": false
128
+ },
129
+ "repeats": 1,
130
+ "should_decontaminate": false,
131
+ "metadata": {
132
+ "version": 1.0,
133
+ "model": "student",
134
+ "base_url": "http://127.0.0.1:8420/v1/completions",
135
+ "tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0100",
136
+ "num_concurrent": 48,
137
+ "tokenized_requests": false,
138
+ "max_retries": 3,
139
+ "config_source": "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/mbpp/mbpp.yaml"
140
+ }
141
+ }
142
+ },
143
+ "versions": {
144
+ "humaneval": 1.0,
145
+ "mbpp": 1.0
146
+ },
147
+ "n-shot": {
148
+ "humaneval": 0,
149
+ "mbpp": 3
150
+ },
151
+ "higher_is_better": {
152
+ "humaneval": {
153
+ "pass_at_k": true,
154
+ "pass@1": true
155
+ },
156
+ "mbpp": {
157
+ "pass_at_1": true
158
+ }
159
+ },
160
+ "n-samples": {
161
+ "humaneval": {
162
+ "original": 164,
163
+ "effective": 164
164
+ },
165
+ "mbpp": {
166
+ "original": 500,
167
+ "effective": 500
168
+ }
169
+ },
170
+ "config": {
171
+ "model": "local-completions",
172
+ "model_args": {
173
+ "model": "student",
174
+ "base_url": "http://127.0.0.1:8420/v1/completions",
175
+ "tokenizer": "outputs/healed/grid_general/glean_keep50_s1224/step0100",
176
+ "num_concurrent": 48,
177
+ "tokenized_requests": false,
178
+ "max_retries": 3
179
+ },
180
+ "batch_size": 1,
181
+ "batch_sizes": [],
182
+ "device": "cuda:0",
183
+ "use_cache": null,
184
+ "limit": null,
185
+ "bootstrap_iters": 100000,
186
+ "gen_kwargs": {},
187
+ "random_seed": 0,
188
+ "numpy_seed": 1234,
189
+ "torch_seed": 1234,
190
+ "fewshot_seed": 1234
191
+ },
192
+ "git_hash": "4cd474a",
193
+ "date": 1784395569.954176,
194
+ "pretty_env_info": "PyTorch version: 2.12.1+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Pop!_OS 22.04 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.22.1\nLibc version: glibc-2.35\n\nPython version: 3.12.12 (main, Oct 31 2025, 23:02:31) [Clang 21.1.4 ] (64-bit runtime)\nPython platform: Linux-7.0.11-76070011-generic-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: \nGPU 0: NVIDIA GeForce RTX 3090\nGPU 1: NVIDIA GeForce RTX 3090\nGPU 2: NVIDIA GeForce RTX 3090\n\nNvidia driver version: 580.159.03\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 43 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 48\nOn-line CPU(s) list: 0-47\nVendor ID: AuthenticAMD\nModel name: AMD EPYC 7402P 24-Core Processor\nCPU family: 23\nModel: 49\nThread(s) per core: 2\nCore(s) per socket: 24\nSocket(s): 1\nStepping: 0\nFrequency boost: enabled\nCPU max MHz: 3350.8049\nCPU min MHz: 1500.0000\nBogoMIPS: 5600.11\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid extd_apicid aperfmperf rapl pni pclmulqdq monitor ssse3 fma cx16 sse4_1 sse4_2 movbe popcnt aes xsave avx f16c rdrand lahf_lm cmp_legacy svm extapic cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw ibs skinit wdt tce topoext perfctr_core perfctr_nb bpext perfctr_llc mwaitx cpb cat_l3 cdp_l3 hw_pstate ssbd mba ibrs ibpb stibp vmmcall fsgsbase bmi1 avx2 smep bmi2 cqm rdt_a rdseed adx smap clflushopt clwb sha_ni xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local clzero irperf xsaveerptr rdpru wbnoinvd amd_ppin arat npt lbrv svm_lock nrip_save tsc_scale vmcb_clean flushbyasid decodeassists pausefilter pfthreshold avic v_vmsave_vmload vgif v_spec_ctrl umip rdpid overflow_recov succor smca sev sev_es\nVirtualization: AMD-V\nL1d cache: 768 KiB (24 instances)\nL1i cache: 768 KiB (24 instances)\nL2 cache: 12 MiB (24 instances)\nL3 cache: 128 MiB (8 instances)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0-47\nVulnerability Gather data sampling: Not affected\nVulnerability Ghostwrite: Not affected\nVulnerability Indirect target selection: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Not affected\nVulnerability Old microcode: Not affected\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Mitigation; untrained return thunk; SMT enabled with STIBP protection\nVulnerability Spec rstack overflow: Mitigation; Safe RET\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; IBPB conditional; STIBP always-on; RSB filling; PBRSB-eIBRS Not affected; BHI Not affected\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Not affected\nVulnerability Vmscape: Mitigation; IBPB before exit to userspace\n\nVersions of relevant libraries:\n[pip3] Could not collect\n[conda] Could not collect",
195
+ "transformers_version": "4.57.6",
196
+ "lm_eval_version": "0.4.12",
197
+ "upper_git_hash": null,
198
+ "tokenizer_pad_token": [
199
+ "<pad>",
200
+ "50280"
201
+ ],
202
+ "tokenizer_eos_token": [
203
+ "|||IP_ADDRESS|||",
204
+ "50279"
205
+ ],
206
+ "tokenizer_bos_token": [
207
+ "|||IP_ADDRESS|||",
208
+ "50279"
209
+ ],
210
+ "eot_token_id": 50279,
211
+ "max_length": 2047,
212
+ "task_hashes": {
213
+ "humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40",
214
+ "mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
215
+ },
216
+ "model_source": "local-completions",
217
+ "model_name": "student",
218
+ "model_name_sanitized": "student",
219
+ "system_instruction": null,
220
+ "system_instruction_sha": null,
221
+ "fewshot_as_multiturn": null,
222
+ "chat_template": null,
223
+ "chat_template_sha": null,
224
+ "total_evaluation_time_seconds": "195.39144668500012"
225
+ }