{ "schema_version": 1, "title": "Reproduction: Softmax as Linear Attention in the Large-Prompt Regime: a Measure-based Perspective", "emoji": "🧮", "space_id": "snaykey/repro-tabcascade", "paper": { "arxiv_id": "2512.11784", "openreview_id": "MvuCgK0Qns" }, "tags": [ "icml2026-repro", "paper-MvuCgK0Qns" ], "updated_at": "2026-07-28T00:00:00+00:00", "root": { "slug": "index", "title": "Reproduction: Softmax as Linear Attention in the Large-Prompt Regime: a Measure-based Perspective", "file": "pages/index.md", "children": [ { "slug": "executive-summary", "title": "Executive summary", "file": "pages/executive-summary/page.md", "children": [] }, { "slug": "claim-1-prop-31-output-concentration", "title": "Proposition 3.1 establishes non-asymptotic concentration bounds of order σ⁶ln(L)/L^(c₂/σ²) showing finite-prompt softmax attention outputs converge to their infinite-prompt counterparts as prompt length L grows, for i.i.d. Gaussian inputs (Section 3, Proposition 3.1).", "file": "pages/claim-1-prop-31-output-concentration/page.md", "children": [] }, { "slug": "claim-2-prop-34-gradient-concentration", "title": "Proposition 3.4 extends the concentration result to gradients, giving comparable convergence rates for gradient stability along the training trajectory (Section 3, Proposition 3.4).", "file": "pages/claim-2-prop-34-gradient-concentration/page.md", "children": [] }, { "slug": "claim-3-lemma-21-affine-reduction", "title": "Lemma 2.1 shows that under Gaussian input measures, the infinite-prompt softmax attention operator reduces to an affine map T^(U,V)[μ](z) = Vm + VΓK^⊤Qz, i.e., structurally equivalent to linear attention (Section 2, Lemma 2.1).", "file": "pages/claim-3-lemma-21-affine-reduction/page.md", "children": [] }, { "slug": "claim-4-thm-43-risk-transfer", "title": "Theorem 4.3 proves that the asymptotic training risk of finite-prompt softmax attention under gradient flow is bounded by the infinite-prompt linear-attention risk plus an error term ε, allowing optimization analyses to transfer from linear to softmax attention (Section 4, Theorem 4.3).", "file": "pages/claim-4-thm-43-risk-transfer/page.md", "children": [] }, { "slug": "claim-5-thm-51-bayes-optimal", "title": "Theorem 5.1 shows that in the large-prompt regime, softmax attention achieves Bayes-optimal risk for in-context linear regression with anisotropic Gaussian covariates, extending prior results restricted to the isotropic case (Section 5, Theorem 5.1).", "file": "pages/claim-5-thm-51-bayes-optimal/page.md", "children": [] }, { "slug": "conclusion", "title": "Conclusion", "file": "pages/conclusion/page.md", "children": [] } ] } }