File size: 3,029 Bytes
e2325ae
 
a8d665a
 
e2325ae
 
a8d665a
 
e2325ae
 
 
a8d665a
e2325ae
a8d665a
e2325ae
 
a8d665a
e2325ae
 
 
 
 
 
 
 
 
a8d665a
 
 
e2325ae
 
 
a8d665a
 
 
e2325ae
 
 
a8d665a
 
 
e2325ae
 
 
a8d665a
 
 
 
 
 
 
 
 
e2325ae
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
{
  "schema_version": 1,
  "title": "Reproduction: Softmax as Linear Attention in the Large-Prompt Regime: a Measure-based Perspective",
  "emoji": "🧮",
  "space_id": "snaykey/repro-tabcascade",
  "paper": {
    "arxiv_id": "2512.11784",
    "openreview_id": "MvuCgK0Qns"
  },
  "tags": [
    "icml2026-repro",
    "paper-MvuCgK0Qns"
  ],
  "updated_at": "2026-07-28T00:00:00+00:00",
  "root": {
    "slug": "index",
    "title": "Reproduction: Softmax as Linear Attention in the Large-Prompt Regime: a Measure-based Perspective",
    "file": "pages/index.md",
    "children": [
      {
        "slug": "executive-summary",
        "title": "Executive summary",
        "file": "pages/executive-summary/page.md",
        "children": []
      },
      {
        "slug": "claim-1-prop-31-output-concentration",
        "title": "Proposition 3.1 establishes non-asymptotic concentration bounds of order σ⁶ln(L)/L^(c₂/σ²) showing finite-prompt softmax attention outputs converge to their infinite-prompt counterparts as prompt length L grows, for i.i.d. Gaussian inputs (Section 3, Proposition 3.1).",
        "file": "pages/claim-1-prop-31-output-concentration/page.md",
        "children": []
      },
      {
        "slug": "claim-2-prop-34-gradient-concentration",
        "title": "Proposition 3.4 extends the concentration result to gradients, giving comparable convergence rates for gradient stability along the training trajectory (Section 3, Proposition 3.4).",
        "file": "pages/claim-2-prop-34-gradient-concentration/page.md",
        "children": []
      },
      {
        "slug": "claim-3-lemma-21-affine-reduction",
        "title": "Lemma 2.1 shows that under Gaussian input measures, the infinite-prompt softmax attention operator reduces to an affine map T^(U,V)[μ](z) = Vm + VΓK^⊤Qz, i.e., structurally equivalent to linear attention (Section 2, Lemma 2.1).",
        "file": "pages/claim-3-lemma-21-affine-reduction/page.md",
        "children": []
      },
      {
        "slug": "claim-4-thm-43-risk-transfer",
        "title": "Theorem 4.3 proves that the asymptotic training risk of finite-prompt softmax attention under gradient flow is bounded by the infinite-prompt linear-attention risk plus an error term ε, allowing optimization analyses to transfer from linear to softmax attention (Section 4, Theorem 4.3).",
        "file": "pages/claim-4-thm-43-risk-transfer/page.md",
        "children": []
      },
      {
        "slug": "claim-5-thm-51-bayes-optimal",
        "title": "Theorem 5.1 shows that in the large-prompt regime, softmax attention achieves Bayes-optimal risk for in-context linear regression with anisotropic Gaussian covariates, extending prior results restricted to the isotropic case (Section 5, Theorem 5.1).",
        "file": "pages/claim-5-thm-51-bayes-optimal/page.md",
        "children": []
      },
      {
        "slug": "conclusion",
        "title": "Conclusion",
        "file": "pages/conclusion/page.md",
        "children": []
      }
    ]
  }
}