| { |
| "checks": [ |
| { |
| "check": "validated tasks", |
| "value": 957, |
| "criterion": "= 957", |
| "passed": true |
| }, |
| { |
| "check": "patch-size ratio", |
| "value": 5.277777777777778, |
| "criterion": "within 0.1 of reported 5.2x", |
| "passed": true |
| }, |
| { |
| "check": "task-workload pairs", |
| "value": 253222.2, |
| "criterion": "> 250,000", |
| "passed": true |
| }, |
| { |
| "check": "validation p threshold", |
| "value": 0.002, |
| "criterion": "< 0.01", |
| "passed": true |
| } |
| ], |
| "scope": "Independent arithmetic audit of the reported dataset funnel and benchmark summary; agent inference was not rerun.", |
| "paper_id": "WArbqRUsAe", |
| "title": "FormulaCode: Evaluating Agentic Optimization on Large Codebases", |
| "seed": 1082026, |
| "executed_at": "2026-08-01T09:33:34.772522+00:00", |
| "all_checks_passed": true, |
| "environment": { |
| "python": "3.10.12", |
| "numpy": "1.24.4", |
| "scipy": "1.14.0", |
| "platform": "Linux-5.15.0-139-generic-x86_64-with-glibc2.35" |
| }, |
| "reference_evidence": { |
| "space": "Srishti280992/repro-formulacode-evaluating-agentic-optimization-on-large-codebases", |
| "sha": "353c018b3843ae98dd37e06c841202cdf2f80ee3", |
| "relationship": "separately attributed public reference" |
| } |
| } |
|
|