# /// script # requires-python = ">=3.11" # dependencies = [ # "vllm>=0.19", # "datasets", # "huggingface_hub", # "hf_transfer", # ] # /// """Benchmark a Qwen3.6-27B variant (base or grug) with vLLM. Probes: 1. HumanEval (164) exec pass@1 2. MBPP sanitized (100) exec pass@1 3. GSM8K test (200) numeric match 4. Agentic replay (120) tool-call validity + tool-choice + args validity 5. Repetition stress think-stripped agent replays, long-form, greedy Everywhere: think vs answer token split, grug dialect metrics, loop metrics. Env: HF_TOKEN, VARIANT (base|grug), MODEL_PATH, HUB_REPO (results dest) """ import json import os import re import subprocess import sys import tempfile import time import traceback os.environ.setdefault("HF_HUB_ENABLE_HF_TRANSFER", "1") os.environ.setdefault("VLLM_USE_V1", "1") from huggingface_hub import HfApi, hf_hub_download VARIANT = os.environ.get("VARIANT", "base") MODEL_PATH = os.environ.get("MODEL_PATH", "Qwen/Qwen3.6-27B") HUB_REPO = os.environ.get("HUB_REPO", "ProCreations/grug-27b-lora") api = HfApi() THINK_SPLIT = re.compile(r"", re.I) CODE_BLOCK = re.compile(r"```(?:python)?\n(.*?)```", re.S) TOOL_CALL = re.compile(r"\s*(.*?)\s*", re.S) PARAM = re.compile(r"", re.S) FUNCTION_WORDS = set("the a an is are was were be been being to of and or i we me my " "let's lets that this it will would should could do does did have " "has had am so then just really very can may might".split()) BANNED_THINK = re.compile(r"\b(let me|let's|the user|user wants|i will|i'll|we need|" r"i need|okay so|first, i)\b", re.I) def split_think(text): parts = THINK_SPLIT.split(text, maxsplit=1) if len(parts) == 2: return parts[0].replace("", "").strip(), parts[1].strip() return text.strip(), "" # never closed think def grug_metrics(think): words = re.findall(r"[A-Za-z']+", think.lower()) if not words: return {"fw_ratio": 0.0, "banned": 0, "words": 0} fw = sum(1 for w in words if w in FUNCTION_WORDS) return {"fw_ratio": round(fw / len(words), 4), "banned": len(BANNED_THINK.findall(think)), "words": len(words)} def loop_metrics(text): """Detect degenerate repetition: longest consecutive repeat of any 8-40 word phrase in the tail, and repeated-line ratio.""" words = text.split() worst = 0 tail = words[-1200:] for k in (8, 15, 30): i = len(tail) - k while i >= k: reps = 1 j = i while j >= k and tail[j - k:j] == tail[i:i + k]: reps += 1 j -= k worst = max(worst, reps if reps > 1 else 0) i -= k if reps > 1 else 1 lines = [l for l in text.splitlines() if l.strip()] dup = 0 if len(lines) > 10: from collections import Counter c = Counter(lines) dup = sum(v for v in c.values() if v >= 3) / len(lines) return {"max_phrase_reps": worst, "dup_line_ratio": round(dup, 3)} def summarize(name, outs, passed, total, extra=None): toks = sorted(o["tokens"] for o in outs) or [0] thinks = sorted(o["think_tokens"] for o in outs) or [0] fw = [o["grug"]["fw_ratio"] for o in outs if o["grug"]["words"] > 5] s = {"benchmark": name, "n": total, "score_pct": round(100 * passed / max(total, 1), 1), "tokens_mean": round(sum(toks) / len(toks), 1), "tokens_p50": toks[len(toks) // 2], "think_tokens_mean": round(sum(thinks) / len(thinks), 1), "think_tokens_p50": thinks[len(thinks) // 2], "think_fw_ratio_mean": round(sum(fw) / max(len(fw), 1), 4), "think_banned_total": sum(o["grug"]["banned"] for o in outs), "answer_fw_ratio_mean": round(sum(grug_metrics(o["answer"])["fw_ratio"] for o in outs if len(o["answer"].split()) > 10) / max(sum(1 for o in outs if len(o["answer"].split()) > 10), 1), 4), "no_eos_pct": round(100 * sum(1 for o in outs if not o["stopped"]) / max(len(outs), 1), 1), "loop_reps_max": max((o["loop"]["max_phrase_reps"] for o in outs), default=0), "loop_flagged_pct": round(100 * sum(1 for o in outs if o["loop"]["max_phrase_reps"] >= 3) / max(len(outs), 1), 1)} if extra: s.update(extra) print(name, json.dumps(s), flush=True) return s def run_python(program, timeout=10): with tempfile.NamedTemporaryFile("w", suffix=".py", delete=False) as f: f.write(program) path = f.name try: r = subprocess.run([sys.executable, path], capture_output=True, timeout=timeout) return r.returncode == 0 except Exception: return False finally: os.unlink(path) def extract_code(answer, fallback_prompt="", entry_point=""): blocks = CODE_BLOCK.findall(answer) code = blocks[-1] if blocks else answer if entry_point and f"def {entry_point}" not in code and fallback_prompt: code = fallback_prompt + "\n" + code return code class Runner: def __init__(self): from vllm import LLM from transformers import AutoTokenizer self.tok = AutoTokenizer.from_pretrained(MODEL_PATH) self.llm = LLM(model=MODEL_PATH, max_model_len=16384, max_num_seqs=200, gpu_memory_utilization=0.92, trust_remote_code=True) def chat(self, prompts, max_new, temperature=0.6, greedy=False): """prompts: list of (messages, tools) -> list of out dicts.""" from vllm import SamplingParams texts = [] for msgs, tools in prompts: texts.append(self.tok.apply_chat_template( msgs, tools=tools if tools else None, tokenize=False, add_generation_prompt=True)) sp = SamplingParams( temperature=0.0 if greedy else temperature, top_p=1.0 if greedy else 0.95, top_k=-1 if greedy else 20, max_tokens=max_new, seed=42) outs = [] for i, o in enumerate(self.llm.generate(texts, sp)): text = o.outputs[0].text think, answer = split_think(text) outs.append({"text": text, "think": think, "answer": answer, "tokens": len(o.outputs[0].token_ids), "think_tokens": len(self.tok(think, add_special_tokens=False)["input_ids"]) if think else 0, "stopped": o.outputs[0].finish_reason == "stop", "grug": grug_metrics(think), "loop": loop_metrics(text)}) return outs def eval_humaneval(rn): from datasets import load_dataset ds = load_dataset("openai/openai_humaneval", split="test") sysp = os.environ.get("SYS_PROMPT") def wrap(msgs): return ([{"role": "system", "content": sysp}] + msgs) if sysp else msgs prompts = [(wrap([{"role": "user", "content": "Complete this Python function. Reply with the FULL working function " f"in a single ```python code block.\n\n```python\n{r['prompt']}\n```"}]), None) for r in ds] outs = rn.chat(prompts, 8192) passed = 0 for r, o in zip(ds, outs): code = extract_code(o["answer"] or o["text"], r["prompt"], r["entry_point"]) o["pass"] = run_python(code + "\n\n" + r["test"] + f"\n\ncheck({r['entry_point']})\n") passed += o["pass"] return summarize("humaneval", outs, passed, len(ds)), outs def eval_mbpp(rn): from datasets import load_dataset ds = load_dataset("google-research-datasets/mbpp", "sanitized", split="test") ds = ds.select(range(min(100, len(ds)))) prompts = [([{"role": "user", "content": f"{r['prompt']}\n\nWrite the solution as a single ```python code block. " f"It must pass tests like:\n{r['test_list'][0]}"}], None) for r in ds] outs = rn.chat(prompts, 6144) passed = 0 for r, o in zip(ds, outs): code = extract_code(o["answer"] or o["text"]) prog = "\n".join(r.get("test_imports") or []) + "\n" + code + "\n" + "\n".join(r["test_list"]) o["pass"] = run_python(prog) passed += o["pass"] return summarize("mbpp", outs, passed, len(ds)), outs def eval_gsm8k(rn): from datasets import load_dataset ds = load_dataset("openai/gsm8k", "main", split="test").select(range(200)) prompts = [([{"role": "user", "content": r["question"] + "\n\nGive the final numeric answer on the last line as: " "#### "}], None) for r in ds] outs = rn.chat(prompts, 4096) passed = 0 for r, o in zip(ds, outs): ref = r["answer"].split("####")[-1].strip().replace(",", "") text = (o["answer"] or o["text"]).replace(",", "") m = re.findall(r"####\s*(-?[\d.]+)", text) or re.findall(r"(-?[\d.]+)", text) try: o["pass"] = bool(m) and abs(float(m[-1]) - float(ref)) < 1e-6 except Exception: o["pass"] = False passed += o["pass"] return summarize("gsm8k", outs, passed, len(ds)), outs def _norm_args_ev(a): if isinstance(a, str): try: a = json.loads(a) except Exception: a = {"raw": a} if a is None: a = {} if not isinstance(a, dict): a = {"arguments": a if isinstance(a, str) else json.dumps(a, ensure_ascii=False)} return a def _strip_history_thinks(msgs): out = [] for m in msgs: m2 = dict(m) c = m2.get("content") or "" if m2["role"] == "assistant" and "" in c: m2["content"] = c.split("")[-1].lstrip("\n") out.append(m2) return out def _load_heldout(): path = hf_hub_download("ProCreations/grug-think", "heldout/heldout_agentic.jsonl", repo_type="dataset") return [json.loads(l) for l in open(path)] def eval_agentic(rn): exs = _load_heldout() prompts, refs = [], [] for ex in exs: msgs = ex["messages"] call_turns = [i for i, m in enumerate(msgs) if m["role"] == "assistant" and m.get("tool_calls")] if not call_turns: continue idx = call_turns[-1] ctx = _strip_history_thinks(msgs[:idx]) est = sum(len(m.get("content") or "") for m in ctx) // 3 if est > 11000 or not ctx or ctx[-1]["role"] == "assistant": continue # normalize tool_calls in history for the template for m in ctx: if m.get("tool_calls"): m["tool_calls"] = [{"type": "function", "function": { "name": tc["function"]["name"], "arguments": _norm_args_ev(tc["function"]["arguments"])}} for tc in m["tool_calls"]] prompts.append((ctx, ex.get("tools"))) refs.append((msgs[idx]["tool_calls"][0]["function"]["name"], ex.get("tools") or [])) if len(prompts) >= 120: break outs = rn.chat(prompts, 2560) ok = match = args_ok = 0 for (ref_name, tools), o in zip(refs, outs): m = TOOL_CALL.search(o["text"]) o["pass"] = bool(m) if m: ok += 1 name = m.group(1) match += (name == ref_name) schema = next((t["function"] for t in tools if t.get("function", {}).get("name") == name), None) if schema: given = set(PARAM.findall(m.group(2))) props = set((schema.get("parameters", {}).get("properties") or {}).keys()) req = set(schema.get("parameters", {}).get("required") or []) args_ok += (given.issubset(props) if props else 1) and req.issubset(given) n = len(outs) return summarize("agentic_replay", outs, ok, n, {"tool_name_match_pct": round(100 * match / max(n, 1), 1), "args_valid_pct": round(100 * args_ok / max(ok, 1), 1)}), outs def eval_math500(rn): """Unseen surface: HuggingFaceH4/MATH-500 was never used anywhere in grug training data generation - clean anti-contamination probe.""" from datasets import load_dataset ds = load_dataset("HuggingFaceH4/MATH-500", split="test").select(range(150)) prompts = [([{"role": "user", "content": r["problem"] + "\n\nGive the final answer on the last line as: ANSWER: "}], None) for r in ds] outs = rn.chat(prompts, int(os.environ.get("MATH500_BUDGET", 4096))) def norm(s): s = str(s).strip().replace(" ", "").replace("\\!", "").replace("\\,", "") s = s.replace("\\left", "").replace("\\right", "").rstrip(".$").lstrip("$") s = s.replace("dfrac", "frac").replace("tfrac", "frac") return s passed = 0 for r, o in zip(ds, outs): text = o["answer"] or o["text"] m = re.findall(r"ANSWER:\s*(.+)", text) got = m[-1].strip() if m else (text.strip().splitlines()[-1] if text.strip() else "") o["pass"] = norm(got) == norm(r["answer"]) or (norm(r["answer"]) in norm(got) and len(norm(r["answer"])) > 0) passed += o["pass"] return summarize("math500", outs, passed, len(ds)), outs def eval_repetition(rn): exs = _load_heldout() # a) agent-style replay deep into think-stripped trajectories, GREEDY prompts = [] for ex in exs: msgs = ex["messages"] a_turns = [i for i, m in enumerate(msgs) if m["role"] == "assistant"] if len(a_turns) < 4: continue idx = a_turns[int(len(a_turns) * 0.7)] ctx = _strip_history_thinks(msgs[:idx]) if ctx[-1]["role"] == "assistant": ctx = ctx[:-1] est = sum(len(m.get("content") or "") for m in ctx) // 3 if est > 11000: continue for m in ctx: if m.get("tool_calls"): m["tool_calls"] = [{"type": "function", "function": { "name": tc["function"]["name"], "arguments": _norm_args_ev(tc["function"]["arguments"])}} for tc in m["tool_calls"]] prompts.append((ctx, ex.get("tools"))) if len(prompts) >= 25: break outs_a = rn.chat(prompts, 3072, greedy=True) # b) long-form generation, greedy (worst case for loops) topics = ["writing a detailed 1500-word beginner guide to training LoRA adapters", "explaining how a garbage collector works, in depth with examples", "a detailed comparison of 6 sorting algorithms with complexity analysis", "designing a REST API for a library system, all endpoints documented", "a step-by-step debugging walkthrough of a race condition in Python", "explaining TCP congestion control in detail", "a long technical blog post about SQLite internals", "documenting a git branching strategy for a 20-person team", "an in-depth explanation of how DNS resolution works end to end", "a thorough code review checklist with rationale for each item"] outs_b = rn.chat([([{"role": "user", "content": f"Write {t}."}], None) for t in topics], 6144, greedy=True) # c) multi-turn chat continuation (history thinks stripped), sampled chats = [] base_qa = [("What's a good way to learn Rust?", "Start with the official book..."), ("How do I read a file in Rust?", "Use std::fs::read_to_string..."), ("What about async?", "Tokio is the standard runtime...")] for i in range(8): msgs = [] for q, a in base_qa: msgs.append({"role": "user", "content": q}) msgs.append({"role": "assistant", "content": a}) msgs.append({"role": "user", "content": ["Now show me a full example combining all of that.", "Summarize everything you told me so far in detail.", "Write a longer tutorial based on this conversation.", "What are 15 common mistakes beginners make with this?"][i % 4]}) chats.append((msgs, None)) outs_c = rn.chat(chats, 4096) outs = outs_a + outs_b + outs_c healthy = sum(1 for o in outs if o["loop"]["max_phrase_reps"] < 3 and o["loop"]["dup_line_ratio"] < 0.2 and (o["stopped"] or o["tokens"] < 6144)) s = summarize("repetition_stress", outs, healthy, len(outs), {"agent_replay_n": len(outs_a), "longform_n": len(outs_b), "chat_n": len(outs_c), "think_closed_pct": round(100 * sum(1 for o in outs if o["answer"] != "" or "" in o["text"]) / max(len(outs), 1), 1)}) return s, outs def main(): t0 = time.time() rn = Runner() res = {"variant": VARIANT, "model": MODEL_PATH} dumps = {} probes = [("humaneval", eval_humaneval), ("mbpp", eval_mbpp), ("gsm8k", eval_gsm8k), ("math500", eval_math500), ("agentic", eval_agentic), ("repetition", eval_repetition)] only = os.environ.get("PROBES") if only: keep = set(only.split(",")) probes = [p for p in probes if p[0] in keep] for name, fn in probes: try: res[name], outs = fn(rn) dumps[name] = outs except Exception: print(traceback.format_exc(), flush=True) res[name] = {"benchmark": name, "error": True} res["wall_minutes"] = round((time.time() - t0) / 60, 1) api.upload_file(path_or_fileobj=json.dumps(res, indent=2).encode(), path_in_repo=f"results/{VARIANT}.json", repo_id=HUB_REPO, repo_type="model") # dump raw generations (truncated) for inspection slim = {k: [{kk: (vv[:2500] if isinstance(vv, str) else vv) for kk, vv in o.items() if kk != "token_ids"} for o in v] for k, v in dumps.items()} api.upload_file(path_or_fileobj=json.dumps(slim).encode(), path_in_repo=f"results/{VARIANT}_generations.json", repo_id=HUB_REPO, repo_type="model") print("EVAL COMPLETE", flush=True) if __name__ == "__main__": try: main() except Exception: api.upload_file(path_or_fileobj=traceback.format_exc().encode(), path_in_repo=f"results/{VARIANT}_CRASHED", repo_id=HUB_REPO, repo_type="model") raise