NEXORA / scripts /stream_benchmark.py
devildasdf's picture
Release validated NEXORA research prototype, tiny weights and evidence
12496fc verified
Raw History Blame Contribute Delete
1.11 kB
from pathlib import Path
import argparse
import json
import sys
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from nexora.inference import HFBackend
def main():
p = argparse.ArgumentParser()
p.add_argument("--model", default=".cache/Qwen3.5-0.8B")
args = p.parse_args()
backend = HFBackend(args.model, max_new_tokens=32)
messages = [{"role": "user", "content": "Describe a queue in one short sentence."}]
nonstream = backend.complete(messages)
chunks = list(backend.stream(messages))
report = {"model": args.model, "prompt": messages[0]["content"], "nonstream": nonstream, "streamed": "".join(chunks),
"stream_matches_nonstream": nonstream == "".join(chunks), "chunks": len(chunks), **backend.last_metrics,
"limitations": "One greedy request, warm model, no concurrent load or audio"}
if not report["stream_matches_nonstream"]:
raise AssertionError("Streaming parity failed")
Path("reports/streaming.json").write_text(json.dumps(report, indent=2))
print(json.dumps(report))
if __name__ == "__main__":
main()