Download scripts/benchmark.py from devildasdf/NEXORA: direct link, hf CLI and curl.
- Browser
- Download file 4.54 kB
-
https://huggingface.co/devildasdf/NEXORA/resolve/main/scripts/benchmark.py
- Command line
-
hf download hf://devildasdf/NEXORA/scripts/benchmark.py
-
curl -L -o benchmark.py https://huggingface.co/devildasdf/NEXORA/resolve/main/scripts/benchmark.py
4.54 kB
| """Measured baseline, tokenizer and agent experiments. All failures are retained.""" | |
| from pathlib import Path | |
| from dataclasses import asdict | |
| import json | |
| import sys | |
| import tempfile | |
| import time | |
| import argparse | |
| sys.path.insert(0, str(Path(__file__).resolve().parents[1])) | |
| from nexora.inference import HFBackend | |
| from nexora.tokenizer import ByteTokenizer | |
| from nexora.agent import Agent | |
| from nexora.tools import Executor, Policy | |
| from nexora.evaluation import wilson, percentiles | |
| SAMPLES = { | |
| "english": "A reliable assistant inspects evidence before claiming that a task succeeded.", | |
| "hindi": "सहायक को कार्य पूरा होने का दावा करने से पहले प्रमाण की जाँच करनी चाहिए।", | |
| "hinglish": "Pehle code run karo aur tests ka output check karo, phir result batao.", | |
| "python": "def total(values):\n return sum(x for x in values if x is not None)\n", | |
| "json": '{"tool":"filesystem.read","arguments":{"path":"src/main.py"}}', | |
| "math": "For x ∈ ℝ, x² + 2x + 1 = (x + 1)²; ∑ᵢ xᵢ / n.", | |
| "shell": "git diff --stat && python -m pytest -q", | |
| "xml_url": '<source href="https://example.org/api?q=a%20b">Unicode: λ</source>', | |
| } | |
| def main(): | |
| parser = argparse.ArgumentParser() | |
| parser.add_argument("--model", default=".cache/Qwen3-0.6B") | |
| parser.add_argument("--report", default="baseline") | |
| args = parser.parse_args() | |
| out = Path("reports") | |
| out.mkdir(exist_ok=True) | |
| backend = HFBackend(args.model, max_new_tokens=96) | |
| byte_report = ByteTokenizer().benchmark(SAMPLES) | |
| bpe = {} | |
| for name, text in SAMPLES.items(): | |
| ids = backend.tokenizer.encode(text, add_special_tokens=False) | |
| bpe[name] = {"tokens": len(ids), "characters_per_token": len(text)/len(ids), "roundtrip": backend.tokenizer.decode(ids) == text} | |
| (out / f"tokenizer-{args.report}.json").write_text(json.dumps({"byte": byte_report, "qwen_bpe": bpe, "limitation": "Eight public examples, not a representative benchmark or vocabulary-size study"}, indent=2, ensure_ascii=False), encoding="utf-8") | |
| cases = [ | |
| ("arithmetic", "Return only the integer: 17 * 23.", lambda s: s.strip() == "391"), | |
| ("json", 'Return only a JSON object with key "ready" and boolean value true.', lambda s: json.loads(s) == {"ready": True}), | |
| ("ordering", "Sort these integers ascending. Return only a JSON array: 19, -3, 7, 0.", lambda s: json.loads(s) == [-3, 0, 7, 19]), | |
| ("reasoning", "Every raven is a bird. Some birds swim. Must every raven swim? Answer only yes or no.", lambda s: s.strip().lower().rstrip(".") == "no"), | |
| ] | |
| results = [] | |
| for name, prompt, verify in cases: | |
| answer = backend.complete([{"role": "user", "content": prompt}]) | |
| try: | |
| passed = verify(answer) is True | |
| except Exception: | |
| passed = False | |
| row = {"category": name, "prompt": prompt, "answer": answer, "passed": passed, **backend.last_metrics} | |
| results.append(row) | |
| print(json.dumps(row), flush=True) | |
| with tempfile.TemporaryDirectory(prefix="nexora-agent-") as folder: | |
| root = Path(folder) | |
| (root / "note.txt").write_text("The test project uses a bounded queue.") | |
| executor = Executor(Policy(folder, permissions=["READ", "WRITE"])) | |
| task = "Read note.txt, then create summary.txt containing exactly: bounded queue. Finish after the file is written." | |
| result = Agent(backend, executor, verifier=lambda: (root / "summary.txt").is_file() and (root / "summary.txt").read_text() == "bounded queue", max_steps=6).run(task) | |
| n, successes = len(results), sum(r["passed"] for r in results) | |
| report = {"model": args.model, "total_parameters": sum(p.numel() for p in backend.model.parameters()), "modified_weights": False, "device": "cpu", "cases": results, | |
| "pass_at_1": successes/n, "wilson_95": wilson(successes, n), | |
| "latency_seconds": percentiles([r["seconds"] for r in results]), "agent": result, | |
| "limitations": ["Public hand-authored smoke tests; contamination unknown", "Four cases cannot rank models", "Agent file task is not repository-level coding evaluation", "Non-streaming latency only", "No private data used"]} | |
| (out / f"{args.report}.json").write_text(json.dumps(report, indent=2, ensure_ascii=False), encoding="utf-8") | |
| print(json.dumps({"pass_at_1": successes/n, "agent_status": result["status"]})) | |
| if __name__ == "__main__": | |
| main() | |