| |
| """HumanEval pass@1 — a coding guardrail (did security tuning help or hurt general coding?). |
| |
| Loads a model in-process (HF generate), prompts each HumanEval problem, extracts the |
| generated function, executes it against the official unit tests in a subprocess, and |
| reports pass@1. Emits reports/eval/<label>_coding.{json,md} in the same shape as the |
| other evals so base vs merged is directly comparable, and includes published reference |
| numbers for context. |
| |
| Inference is forward-only (unaffected by any backward-kernel issue). Generated code is |
| executed in a subprocess with a timeout on a disposable GPU VM. |
| |
| Usage: |
| python training/scripts/eval_coding.py --model Qwen/Qwen3.6-27B --label base_a100 |
| python training/scripts/eval_coding.py --model /workspace/checkpoints/qwen36_a100_stage1_merged --label merged |
| """ |
|
|
| from __future__ import annotations |
|
|
| import argparse |
| import json |
| import re |
| import subprocess |
| import sys |
| import tempfile |
| from pathlib import Path |
|
|
| sys.path.insert(0, str(Path(__file__).resolve().parent)) |
| from eval_hf_model import load_model |
| from intraining_eval import _generate |
|
|
| |
| REFERENCE = { |
| "Qwen2.5-Coder-32B-Instruct": 0.92, |
| "GPT-4o": 0.90, |
| "DeepSeek-V3": 0.91, |
| "Llama-3.1-70B-Instruct": 0.80, |
| "Qwen2.5-7B-Instruct": 0.84, |
| "CodeLlama-34B": 0.51, |
| } |
|
|
| CODE_BLOCK = re.compile(r"```(?:python)?\s*(.*?)```", re.DOTALL | re.IGNORECASE) |
|
|
|
|
| def extract_code(text: str, entry_point: str, prompt: str) -> str: |
| body = text.rsplit("</think>", 1)[-1] if "</think>" in text else text |
| m = CODE_BLOCK.search(body) |
| code = m.group(1).strip() if m else body.strip() |
| |
| |
| if f"def {entry_point}" in code: |
| return code |
| return prompt + "\n" + code |
|
|
|
|
| def run_program(program: str, timeout: int) -> bool: |
| with tempfile.NamedTemporaryFile("w", suffix=".py", delete=False) as f: |
| f.write(program) |
| path = f.name |
| try: |
| proc = subprocess.run([sys.executable, path], capture_output=True, timeout=timeout) |
| return proc.returncode == 0 |
| except Exception: |
| return False |
| finally: |
| try: |
| Path(path).unlink() |
| except OSError: |
| pass |
|
|
|
|
| def parse_args() -> argparse.Namespace: |
| p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) |
| p.add_argument("--model", required=True) |
| p.add_argument("--label", required=True) |
| p.add_argument("--report-dir", default="reports/eval") |
| p.add_argument("--limit", type=int, default=0, help="0 = all 164 problems.") |
| p.add_argument("--max-new-tokens", type=int, default=640) |
| p.add_argument("--dtype", default="bfloat16") |
| p.add_argument("--device-map", default="auto") |
| p.add_argument("--timeout", type=int, default=12, help="Per-problem execution timeout (s).") |
| return p.parse_args() |
|
|
|
|
| def main() -> int: |
| args = parse_args() |
| from datasets import load_dataset |
| from transformers import AutoTokenizer |
|
|
| ds = load_dataset("openai/openai_humaneval", split="test") |
| if args.limit: |
| ds = ds.select(range(min(args.limit, len(ds)))) |
|
|
| tokenizer = AutoTokenizer.from_pretrained(args.model, trust_remote_code=True) |
| if tokenizer.pad_token_id is None: |
| tokenizer.pad_token = tokenizer.eos_token |
| model = load_model(args.model, args.dtype, args.device_map) |
|
|
| sys_msg = ( |
| "You are an expert Python programmer. Complete the function. Respond with a single " |
| "```python code block containing the full function definition and any needed imports, " |
| "and no explanation." |
| ) |
| passed = 0 |
| total = len(ds) |
| for i, ex in enumerate(ds): |
| messages = [{"role": "system", "content": sys_msg}, {"role": "user", "content": ex["prompt"]}] |
| out = _generate(model, tokenizer, messages, args.max_new_tokens, enable_thinking=False) |
| code = extract_code(out, ex["entry_point"], ex["prompt"]) |
| program = code + "\n\n" + ex["test"] + f"\n\ncheck({ex['entry_point']})\n" |
| if run_program(program, args.timeout): |
| passed += 1 |
| if (i + 1) % 20 == 0: |
| print(f"[coding] {i+1}/{total} pass@1 so far {passed/(i+1):.3f}", flush=True) |
|
|
| pass_at_1 = passed / total if total else 0.0 |
| result = {"kind": "humaneval", "n": total, "pass@1": round(pass_at_1, 4), "passed": passed} |
| report_dir = Path(args.report_dir) |
| report_dir.mkdir(parents=True, exist_ok=True) |
| payload = {"label": args.label, "model": args.model, "results": [result], "reference_pass@1": REFERENCE} |
| (report_dir / f"{args.label}_coding.json").write_text(json.dumps(payload, indent=2), encoding="utf-8") |
|
|
| lines = [f"# HumanEval — {args.label}", "", f"- Model: `{args.model}`", |
| f"- pass@1: **{pass_at_1:.2%}** ({passed}/{total})", "", "## Reference pass@1 (published, approx)"] |
| for k, v in REFERENCE.items(): |
| lines.append(f"- {k}: {v:.0%}") |
| (report_dir / f"{args.label}_coding.md").write_text("\n".join(lines) + "\n", encoding="utf-8") |
| print(json.dumps(payload, indent=2)) |
| return 0 |
|
|
|
|
| if __name__ == "__main__": |
| raise SystemExit(main()) |
|
|