{ "checks": [ { "check": "validated tasks", "value": 957, "criterion": "= 957", "passed": true }, { "check": "patch-size ratio", "value": 5.277777777777778, "criterion": "within 0.1 of reported 5.2x", "passed": true }, { "check": "task-workload pairs", "value": 253222.2, "criterion": "> 250,000", "passed": true }, { "check": "validation p threshold", "value": 0.002, "criterion": "< 0.01", "passed": true } ], "scope": "Independent arithmetic audit of the reported dataset funnel and benchmark summary; agent inference was not rerun.", "paper_id": "WArbqRUsAe", "title": "FormulaCode: Evaluating Agentic Optimization on Large Codebases", "seed": 1082026, "executed_at": "2026-08-01T09:33:34.772522+00:00", "all_checks_passed": true, "environment": { "python": "3.10.12", "numpy": "1.24.4", "scipy": "1.14.0", "platform": "Linux-5.15.0-139-generic-x86_64-with-glibc2.35" }, "reference_evidence": { "space": "Srishti280992/repro-formulacode-evaluating-agentic-optimization-on-large-codebases", "sha": "353c018b3843ae98dd37e06c841202cdf2f80ee3", "relationship": "separately attributed public reference" } }