#!/usr/bin/env python3 """ Benchmark: run pipeline on N copies of a PDF and collect timing + score metrics. Usage: python scripts/benchmark_agents.py path/to/paper.pdf --runs 5 """ import argparse import asyncio import json import sys import time from pathlib import Path sys.path.insert(0, str(Path(__file__).parent.parent)) from dotenv import load_dotenv load_dotenv() from orchestrators.custom_orchestrator import CitationEdgeOrchestrator from utils.helpers import generate_id async def run_once(orch, pdf_path, run_num): job_id = generate_id(f"bench{run_num}") t0 = time.monotonic() try: result = await orch.run(job_id=job_id, pdf_path=str(pdf_path)) duration = time.monotonic() - t0 scores = result.get("scores", {}) return { "run": run_num, "job_id": job_id, "status": result.get("status"), "duration_s": round(duration, 2), "overall_score": scores.get("overall_score", 0), "literary_score": scores.get("literary_score", 0), "argument_score": scores.get("argument_score", 0), "agents": {k: v.get("duration_s") for k, v in result.get("agents", {}).items()}, } except Exception as e: return {"run": run_num, "status": "error", "error": str(e)} async def main(): parser = argparse.ArgumentParser(description="Benchmark CitationEdge pipeline") parser.add_argument("pdf_path", help="Path to the research paper PDF") parser.add_argument("--runs", type=int, default=3, help="Number of runs (default: 3)") args = parser.parse_args() pdf_path = Path(args.pdf_path) if not pdf_path.exists(): print(f"Error: PDF not found: {pdf_path}") sys.exit(1) print(f"Benchmarking {args.runs} runs on: {pdf_path}\n") orch = CitationEdgeOrchestrator() results = [] for i in range(1, args.runs + 1): print(f"Run {i}/{args.runs}...", end=" ", flush=True) r = await run_once(orch, pdf_path, i) print(f"{r.get('duration_s', '?')}s | score={r.get('overall_score', '?')}") results.append(r) # Summary durations = [r["duration_s"] for r in results if "duration_s" in r] scores = [r["overall_score"] for r in results if "overall_score" in r] summary = { "runs": args.runs, "avg_duration_s": round(sum(durations) / len(durations), 2) if durations else 0, "avg_score": round(sum(scores) / len(scores), 2) if scores else 0, "results": results, } output_path = Path("data/benchmark_results.json") output_path.parent.mkdir(exist_ok=True) with open(output_path, "w") as f: json.dump(summary, f, indent=2) print(f"\nBenchmark complete. Avg duration: {summary['avg_duration_s']}s | Avg score: {summary['avg_score']}/10") print(f"Results saved to {output_path}") if __name__ == "__main__": asyncio.run(main())