Buckets:
| { | |
| "schema_version": 1, | |
| "title": "Reproduction: Real-Time Visual Attribution Streaming in Thinking Model", | |
| "emoji": "🎯", | |
| "space_id": null, | |
| "paper": { | |
| "arxiv_id": "2604.16587" | |
| }, | |
| "tags": [ | |
| "icml2026-repro", | |
| "paper-eVr10aZZIw" | |
| ], | |
| "updated_at": "2026-07-20T05:44:09+00:00", | |
| "root": { | |
| "slug": "index", | |
| "title": "Reproduction: Real-Time Visual Attribution Streaming in Thinking Model", | |
| "file": "pages/index.md", | |
| "children": [ | |
| { | |
| "slug": "executive-summary", | |
| "title": "Executive summary", | |
| "file": "pages/executive-summary/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-1-vstream-trains-a-lightweight-linear-estimator-with-only-l-h-parameters-e", | |
| "title": "Claim 1: vStream trains a lightweight linear estimator with only L×H parameters (e.g., 1,152 weights for 32 layers x 36 heads) to predict visual-region ablation effects from cached cross-attention features, avoiding costly backward passes or perturbation-based attribution (Section on method).", | |
| "file": "pages/claim-1-vstream-trains-a-lightweight-linear-estimator-with-only-l-h-parameters-e/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-2-vstream-achieves-a-linear-datamodeling-score-lds-of-approximately-0-70-0", | |
| "title": "Claim 2: vStream achieves a Linear Datamodeling Score (LDS) of approximately 0.70-0.72, matching gradient-based baselines such as AttnLRP (LDS 0.69-0.71), while running at 0.024 sec per 10 tokens versus 2.80 sec per 10 tokens for InputGrad, roughly a 117x speedup (Table 1).", | |
| "file": "pages/claim-2-vstream-achieves-a-linear-datamodeling-score-lds-of-approximately-0-70-0/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-3-the-method-is-evaluated-across-four-thinking-vision-language-models-qwen", | |
| "title": "Claim 3: The method is evaluated across four thinking vision-language models (Qwen3-VL-8B-Thinking, GLM-4.1V-9B-Thinking, MiMo-VL-7B, Cosmos-Reason1-7B) and five task categories including MathVista, MathVision, ScienceQA, DocVQA, and GQA (Section on experimental setup).", | |
| "file": "pages/claim-3-the-method-is-evaluated-across-four-thinking-vision-language-models-qwen/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-4-an-estimator-trained-on-one-task-category-retains-75-90-of-its-attributi", | |
| "title": "Claim 4: An estimator trained on one task category retains 75-90% of its attribution performance when transferred to other categories, with strongest Math-to-Science transfer (LDS 0.62-0.63) and weakest transfer to Document tasks (LDS 0.54-0.58) (Table 2).", | |
| "file": "pages/claim-4-an-estimator-trained-on-one-task-category-retains-75-90-of-its-attributi/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-5-attribution-space-trajectory-analysis-shows-successful-reasoning-chains", | |
| "title": "Claim 5: Attribution-space trajectory analysis shows successful reasoning chains have shorter path length (0.003 vs. 0.006) and lower tortuosity (13.7 vs. 25.4) than failed chains, enabling early failure prediction with AUC 0.69 at only 30% of reasoning completion (Section on trajectory analysis).", | |
| "file": "pages/claim-5-attribution-space-trajectory-analysis-shows-successful-reasoning-chains/page.md", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "conclusion", | |
| "title": "Conclusion", | |
| "file": "pages/conclusion/page.md", | |
| "children": [] | |
| } | |
| ] | |
| }, | |
| "agent_view_tokens": 2520, | |
| "revision": "1784526249285331706" | |
| } |
Xet Storage Details
- Size:
- 3.62 kB
- Xet hash:
- d444d589ad60383a29348a8c7dfeb330ffa6a512f7b6bed6e5b9e9dc40baa988
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.