Spaces:
Running
Running
| { | |
| "schema_version": 1, | |
| "title": "Repro - Quantifying Frontier LLM Capabilities for Container Sandbox Escape", | |
| "emoji": "\ud83d\udd2c", | |
| "space_id": "bertfil/19AbP986bv", | |
| "paper": { | |
| "arxiv_id": "2603.02277", | |
| "openreview_id": "19AbP986bv", | |
| "title": "Quantifying Frontier LLM Capabilities for Container Sandbox Escape", | |
| "url": "https://arxiv.org/abs/2603.02277" | |
| }, | |
| "tags": [ | |
| "icml2026-repro", | |
| "paper-19AbP986bv" | |
| ], | |
| "updated_at": "2026-07-20T03:52:38+00:00", | |
| "root": { | |
| "slug": "index", | |
| "title": "Repro - Quantifying Frontier LLM Capabilities for Container Sandbox Escape", | |
| "file": "pages/index.md", | |
| "children": [ | |
| { | |
| "slug": "00-scored-evidence-summary", | |
| "title": "00 - Scored evidence summary", | |
| "file": "pages/00-scored-evidence-summary/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-1-sandboxescapebench-comprises-18-scenarios-spanning", | |
| "title": "Claim 1 - SandboxEscapeBench comprises 18 scenarios spanning three att\u2026", | |
| "file": "pages/claim-1-sandboxescapebench-comprises-18-scenarios-spanning/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-2-the-benchmark-uses-a-nested-sandbox-in-sandbox-arc", | |
| "title": "Claim 2 - The benchmark uses a nested 'sandbox-in-sandbox' architectur\u2026", | |
| "file": "pages/claim-2-the-benchmark-uses-a-nested-sandbox-in-sandbox-arc/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-3-opus-achieves-an-overall-success-rate-of-0-49-0-39", | |
| "title": "Claim 3 - Opus achieves an overall success rate of 0.49 [0.39, 0.59], \u2026", | |
| "file": "pages/claim-3-opus-achieves-an-overall-success-rate-of-0-49-0-39/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-4-gpt-5-2-performs-significantly-worse-than-gpt-5-on", | |
| "title": "Claim 4 - GPT-5.2 performs significantly worse than GPT-5 on sandbox e\u2026", | |
| "file": "pages/claim-4-gpt-5-2-performs-significantly-worse-than-gpt-5-on/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-5-success-rate-scales-approximately-log-linearly-wit", | |
| "title": "Claim 5 - Success rate scales approximately log-linearly with inferenc\u2026", | |
| "file": "pages/claim-5-success-rate-scales-approximately-log-linearly-wit/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "reproduction-protocol-and-provenance", | |
| "title": "Reproduction protocol and provenance", | |
| "file": "pages/reproduction-protocol-and-provenance/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "conclusion", | |
| "title": "Conclusion", | |
| "file": "pages/conclusion/page.md", | |
| "children": [] | |
| } | |
| ] | |
| }, | |
| "agent_view_tokens": 5290, | |
| "revision": "1784519558619239900" | |
| } |