Spaces:
Running
Running
| { | |
| "schema_version": 1, | |
| "title": "Repro - Maximum Likelihood Reinforcement Learning", | |
| "emoji": "\ud83d\udd2c", | |
| "space_id": "bertfil/EeuLO2BjFN", | |
| "paper": { | |
| "arxiv_id": "2602.02710", | |
| "openreview_id": "EeuLO2BjFN", | |
| "title": "Maximum Likelihood Reinforcement Learning", | |
| "url": "https://arxiv.org/abs/2602.02710" | |
| }, | |
| "tags": [ | |
| "icml2026-repro", | |
| "paper-EeuLO2BjFN" | |
| ], | |
| "updated_at": "2026-07-20T03:52:35+00:00", | |
| "root": { | |
| "slug": "index", | |
| "title": "Repro - Maximum Likelihood Reinforcement Learning", | |
| "file": "pages/index.md", | |
| "children": [ | |
| { | |
| "slug": "00-scored-evidence-summary", | |
| "title": "00 - Scored evidence summary", | |
| "file": "pages/00-scored-evidence-summary/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-1-maxrl-defines-a-compute-indexed-family-of-sample-b", | |
| "title": "Claim 1 - MaxRL defines a compute-indexed family of sample-based objec\u2026", | |
| "file": "pages/claim-1-maxrl-defines-a-compute-indexed-family-of-sample-b/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-2-the-maxrl-objectives-admit-a-simple-unbiased-polic", | |
| "title": "Claim 2 - The MaxRL objectives admit a simple unbiased policy-gradient\u2026", | |
| "file": "pages/claim-2-the-maxrl-objectives-admit-a-simple-unbiased-polic/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-3-the-paper-claims-maxrl-converges-to-maximum-likeli", | |
| "title": "Claim 3 - The paper claims MaxRL converges to maximum-likelihood optim\u2026", | |
| "file": "pages/claim-3-the-paper-claims-maxrl-converges-to-maximum-likeli/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-4-empirically-maxrl-pareto-dominates-tested-existing", | |
| "title": "Claim 4 - Empirically, MaxRL Pareto-dominates tested existing methods \u2026", | |
| "file": "pages/claim-4-empirically-maxrl-pareto-dominates-tested-existing/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-5-maxrl-reports-up-to-20x-test-time-scaling-efficien", | |
| "title": "Claim 5 - MaxRL reports up to 20x test-time scaling efficiency gains c\u2026", | |
| "file": "pages/claim-5-maxrl-reports-up-to-20x-test-time-scaling-efficien/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "reproduction-protocol-and-provenance", | |
| "title": "Reproduction protocol and provenance", | |
| "file": "pages/reproduction-protocol-and-provenance/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "conclusion", | |
| "title": "Conclusion", | |
| "file": "pages/conclusion/page.md", | |
| "children": [] | |
| } | |
| ] | |
| }, | |
| "agent_view_tokens": 8511, | |
| "revision": "1784519555880070100" | |
| } |