Spaces:
Running
Running
Update logbook: Repro: QuArch - A Benchmark for Evaluating LLM Reasoning in Computer Architecture
11be75d verified | { | |
| "schema_version": 1, | |
| "title": "Repro: QuArch - A Benchmark for Evaluating LLM Reasoning in Computer Architecture", | |
| "emoji": "π―", | |
| "space_id": "ParetoOptimal/repro-quarch", | |
| "paper": { | |
| "arxiv_id": "2510.22087" | |
| }, | |
| "tags": [ | |
| "icml2026-repro", | |
| "paper-yU6X1XZl8t" | |
| ], | |
| "updated_at": "2026-07-16T22:16:40+00:00", | |
| "root": { | |
| "slug": "index", | |
| "title": "Repro: QuArch - A Benchmark for Evaluating LLM Reasoning in Computer Architecture", | |
| "file": "pages/index.md", | |
| "children": [ | |
| { | |
| "slug": "paper-approach", | |
| "title": "Paper & approach", | |
| "file": "pages/paper-approach/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "official-claim-1-qa-pairs", | |
| "title": "Official Claim 1 β 2,671 expert-validated QA pairs (composition & coverage)", | |
| "file": "pages/official-claim-1-qa-pairs/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "official-claim-2-frontier-gaps", | |
| "title": "Official Claim 2 β Frontier accuracies 34β73%, wide higher-order gaps", | |
| "file": "pages/official-claim-2-frontier-gaps/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "official-claim-3-finetuning-design", | |
| "title": "Official Claim 3 β Fine-tuning β 1.99Γ area-efficiency, 40% more viable", | |
| "file": "pages/official-claim-3-finetuning-design/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "conclusion", | |
| "title": "Conclusion", | |
| "file": "pages/conclusion/page.md", | |
| "children": [] | |
| } | |
| ] | |
| }, | |
| "agent_view_tokens": 7747, | |
| "revision": "1784240200861629888" | |
| } |