Spaces:
Running
Running
| { | |
| "schema_version": 1, | |
| "title": "Repro — CapBencher: Give Your LLM Benchmark a Built-in Alarm for Test-Set Overfitting (formerly “How Can I Publish My LLM Benchmark Without Giving the True Answers Away?”)", | |
| "emoji": "🎯", | |
| "space_id": "Boopster/oCNT5PcMSQ", | |
| "paper": { | |
| "arxiv_id": "2505.18102" | |
| }, | |
| "tags": [ | |
| "icml2026-repro", | |
| "paper-oCNT5PcMSQ" | |
| ], | |
| "updated_at": "2026-07-17T23:21:31+00:00", | |
| "root": { | |
| "slug": "index", | |
| "title": "Repro — CapBencher: Give Your LLM Benchmark a Built-in Alarm for Test-Set Overfitting (formerly “How Can I Publish My LLM Benchmark Without Giving the True Answers Away?”)", | |
| "file": "pages/index.md", | |
| "children": [ | |
| { | |
| "slug": "claim-a-detecting-deliberate-contamination", | |
| "title": "Official claim 1 — detecting deliberate contamination", | |
| "file": "pages/claim-a-detecting-deliberate-contamination/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-b-the-bayes-accuracy-cap", | |
| "title": "Official claim 2 — the Bayes-accuracy cap", | |
| "file": "pages/claim-b-the-bayes-accuracy-cap/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "conclusion", | |
| "title": "Conclusion", | |
| "file": "pages/conclusion/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "further-evidence-ranking-and-merge", | |
| "title": "Further evidence — BF16 ranking ladder and merge detection", | |
| "file": "pages/further-evidence-ranking-and-merge/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "methods-and-provenance", | |
| "title": "Methods and provenance", | |
| "file": "pages/methods-and-provenance/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "run-history-and-compute", | |
| "title": "Run history and compute", | |
| "file": "pages/run-history-and-compute/page.md", | |
| "children": [] | |
| } | |
| ] | |
| }, | |
| "agent_view_tokens": 19484, | |
| "revision": "1784330491786477000" | |
| } |