infosec-v1 / code /training /scripts /eval_coding.py
adhikjoshi's picture
Super-squash branch 'main' using huggingface_hub
994182c
Raw
History Blame Contribute Delete
5.36 kB
#!/usr/bin/env python3
"""HumanEval pass@1 — a coding guardrail (did security tuning help or hurt general coding?).
Loads a model in-process (HF generate), prompts each HumanEval problem, extracts the
generated function, executes it against the official unit tests in a subprocess, and
reports pass@1. Emits reports/eval/<label>_coding.{json,md} in the same shape as the
other evals so base vs merged is directly comparable, and includes published reference
numbers for context.
Inference is forward-only (unaffected by any backward-kernel issue). Generated code is
executed in a subprocess with a timeout on a disposable GPU VM.
Usage:
python training/scripts/eval_coding.py --model Qwen/Qwen3.6-27B --label base_a100
python training/scripts/eval_coding.py --model /workspace/checkpoints/qwen36_a100_stage1_merged --label merged
"""
from __future__ import annotations
import argparse
import json
import re
import subprocess
import sys
import tempfile
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from eval_hf_model import load_model # noqa: E402
from intraining_eval import _generate # noqa: E402
# Published HumanEval pass@1 (greedy/0-shot, approximate, for context only).
REFERENCE = {
"Qwen2.5-Coder-32B-Instruct": 0.92,
"GPT-4o": 0.90,
"DeepSeek-V3": 0.91,
"Llama-3.1-70B-Instruct": 0.80,
"Qwen2.5-7B-Instruct": 0.84,
"CodeLlama-34B": 0.51,
}
CODE_BLOCK = re.compile(r"```(?:python)?\s*(.*?)```", re.DOTALL | re.IGNORECASE)
def extract_code(text: str, entry_point: str, prompt: str) -> str:
body = text.rsplit("</think>", 1)[-1] if "</think>" in text else text
m = CODE_BLOCK.search(body)
code = m.group(1).strip() if m else body.strip()
# If the model returned the full function, use it; otherwise treat it as the body
# continuation of the given prompt signature.
if f"def {entry_point}" in code:
return code
return prompt + "\n" + code
def run_program(program: str, timeout: int) -> bool:
with tempfile.NamedTemporaryFile("w", suffix=".py", delete=False) as f:
f.write(program)
path = f.name
try:
proc = subprocess.run([sys.executable, path], capture_output=True, timeout=timeout)
return proc.returncode == 0
except Exception:
return False
finally:
try:
Path(path).unlink()
except OSError:
pass
def parse_args() -> argparse.Namespace:
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
p.add_argument("--model", required=True)
p.add_argument("--label", required=True)
p.add_argument("--report-dir", default="reports/eval")
p.add_argument("--limit", type=int, default=0, help="0 = all 164 problems.")
p.add_argument("--max-new-tokens", type=int, default=640)
p.add_argument("--dtype", default="bfloat16")
p.add_argument("--device-map", default="auto")
p.add_argument("--timeout", type=int, default=12, help="Per-problem execution timeout (s).")
return p.parse_args()
def main() -> int:
args = parse_args()
from datasets import load_dataset
from transformers import AutoTokenizer
ds = load_dataset("openai/openai_humaneval", split="test")
if args.limit:
ds = ds.select(range(min(args.limit, len(ds))))
tokenizer = AutoTokenizer.from_pretrained(args.model, trust_remote_code=True)
if tokenizer.pad_token_id is None:
tokenizer.pad_token = tokenizer.eos_token
model = load_model(args.model, args.dtype, args.device_map)
sys_msg = (
"You are an expert Python programmer. Complete the function. Respond with a single "
"```python code block containing the full function definition and any needed imports, "
"and no explanation."
)
passed = 0
total = len(ds)
for i, ex in enumerate(ds):
messages = [{"role": "system", "content": sys_msg}, {"role": "user", "content": ex["prompt"]}]
out = _generate(model, tokenizer, messages, args.max_new_tokens, enable_thinking=False)
code = extract_code(out, ex["entry_point"], ex["prompt"])
program = code + "\n\n" + ex["test"] + f"\n\ncheck({ex['entry_point']})\n"
if run_program(program, args.timeout):
passed += 1
if (i + 1) % 20 == 0:
print(f"[coding] {i+1}/{total} pass@1 so far {passed/(i+1):.3f}", flush=True)
pass_at_1 = passed / total if total else 0.0
result = {"kind": "humaneval", "n": total, "pass@1": round(pass_at_1, 4), "passed": passed}
report_dir = Path(args.report_dir)
report_dir.mkdir(parents=True, exist_ok=True)
payload = {"label": args.label, "model": args.model, "results": [result], "reference_pass@1": REFERENCE}
(report_dir / f"{args.label}_coding.json").write_text(json.dumps(payload, indent=2), encoding="utf-8")
lines = [f"# HumanEval — {args.label}", "", f"- Model: `{args.model}`",
f"- pass@1: **{pass_at_1:.2%}** ({passed}/{total})", "", "## Reference pass@1 (published, approx)"]
for k, v in REFERENCE.items():
lines.append(f"- {k}: {v:.0%}")
(report_dir / f"{args.label}_coding.md").write_text("\n".join(lines) + "\n", encoding="utf-8")
print(json.dumps(payload, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())