NEXORA / scripts /benchmark.py
devildasdf's picture
Release validated NEXORA research prototype, tiny weights and evidence
12496fc verified
Raw History Blame Contribute Delete
4.54 kB
"""Measured baseline, tokenizer and agent experiments. All failures are retained."""
from pathlib import Path
from dataclasses import asdict
import json
import sys
import tempfile
import time
import argparse
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from nexora.inference import HFBackend
from nexora.tokenizer import ByteTokenizer
from nexora.agent import Agent
from nexora.tools import Executor, Policy
from nexora.evaluation import wilson, percentiles
SAMPLES = {
"english": "A reliable assistant inspects evidence before claiming that a task succeeded.",
"hindi": "सहायक को कार्य पूरा होने का दावा करने से पहले प्रमाण की जाँच करनी चाहिए।",
"hinglish": "Pehle code run karo aur tests ka output check karo, phir result batao.",
"python": "def total(values):\n return sum(x for x in values if x is not None)\n",
"json": '{"tool":"filesystem.read","arguments":{"path":"src/main.py"}}',
"math": "For x ∈ ℝ, x² + 2x + 1 = (x + 1)²; ∑ᵢ xᵢ / n.",
"shell": "git diff --stat && python -m pytest -q",
"xml_url": '<source href="https://example.org/api?q=a%20b">Unicode: λ</source>',
}
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--model", default=".cache/Qwen3-0.6B")
parser.add_argument("--report", default="baseline")
args = parser.parse_args()
out = Path("reports")
out.mkdir(exist_ok=True)
backend = HFBackend(args.model, max_new_tokens=96)
byte_report = ByteTokenizer().benchmark(SAMPLES)
bpe = {}
for name, text in SAMPLES.items():
ids = backend.tokenizer.encode(text, add_special_tokens=False)
bpe[name] = {"tokens": len(ids), "characters_per_token": len(text)/len(ids), "roundtrip": backend.tokenizer.decode(ids) == text}
(out / f"tokenizer-{args.report}.json").write_text(json.dumps({"byte": byte_report, "qwen_bpe": bpe, "limitation": "Eight public examples, not a representative benchmark or vocabulary-size study"}, indent=2, ensure_ascii=False), encoding="utf-8")
cases = [
("arithmetic", "Return only the integer: 17 * 23.", lambda s: s.strip() == "391"),
("json", 'Return only a JSON object with key "ready" and boolean value true.', lambda s: json.loads(s) == {"ready": True}),
("ordering", "Sort these integers ascending. Return only a JSON array: 19, -3, 7, 0.", lambda s: json.loads(s) == [-3, 0, 7, 19]),
("reasoning", "Every raven is a bird. Some birds swim. Must every raven swim? Answer only yes or no.", lambda s: s.strip().lower().rstrip(".") == "no"),
]
results = []
for name, prompt, verify in cases:
answer = backend.complete([{"role": "user", "content": prompt}])
try:
passed = verify(answer) is True
except Exception:
passed = False
row = {"category": name, "prompt": prompt, "answer": answer, "passed": passed, **backend.last_metrics}
results.append(row)
print(json.dumps(row), flush=True)
with tempfile.TemporaryDirectory(prefix="nexora-agent-") as folder:
root = Path(folder)
(root / "note.txt").write_text("The test project uses a bounded queue.")
executor = Executor(Policy(folder, permissions=["READ", "WRITE"]))
task = "Read note.txt, then create summary.txt containing exactly: bounded queue. Finish after the file is written."
result = Agent(backend, executor, verifier=lambda: (root / "summary.txt").is_file() and (root / "summary.txt").read_text() == "bounded queue", max_steps=6).run(task)
n, successes = len(results), sum(r["passed"] for r in results)
report = {"model": args.model, "total_parameters": sum(p.numel() for p in backend.model.parameters()), "modified_weights": False, "device": "cpu", "cases": results,
"pass_at_1": successes/n, "wilson_95": wilson(successes, n),
"latency_seconds": percentiles([r["seconds"] for r in results]), "agent": result,
"limitations": ["Public hand-authored smoke tests; contamination unknown", "Four cases cannot rank models", "Agent file task is not repository-level coding evaluation", "Non-streaming latency only", "No private data used"]}
(out / f"{args.report}.json").write_text(json.dumps(report, indent=2, ensure_ascii=False), encoding="utf-8")
print(json.dumps({"pass_at_1": successes/n, "agent_status": result["status"]}))
if __name__ == "__main__":
main()