Spaces:
Running
Running
| { | |
| "schema_version": "1.0", | |
| "title": "MemoryBench: A Benchmark for Memory and Continual Learning in LLM Systems", | |
| "emoji": "M", | |
| "space_id": "snaykey/repro-memorybench", | |
| "paper": { | |
| "arxiv_id": "2510.17281", | |
| "openreview_id": "If4X4W2HWx" | |
| }, | |
| "tags": [ | |
| "icml2026-repro", | |
| "paper-If4X4W2HWx" | |
| ], | |
| "updated_at": "2026-07-29T16:50:47.378234+00:00", | |
| "root": { | |
| "slug": "index", | |
| "title": "MemoryBench: A Benchmark for Memory and Continual Learning in LLM Systems", | |
| "children": [ | |
| { | |
| "slug": "executive-summary", | |
| "title": "Executive summary", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-1-three-module-framework", | |
| "title": "MemoryBench provides a three-module framework with a task provider, user simulator, and performance monitor for testing LLM-system continual learning from feedback logs (Figure 1)", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-2-dataset-coverage", | |
| "title": "MemoryBench covers 11 public datasets across three domains, four task-format categories, and two languages (Table 2)", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-3-memory-feedback-taxonomy", | |
| "title": "The benchmark includes both declarative/procedural memory and explicit/implicit feedback categories absent from prior memory benchmarks (Table 1)", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-4-advanced-memory-vs-rag", | |
| "title": "Off-policy results show that advanced memory systems such as A-Mem, Mem0, and MemoryOS do not consistently outperform simpler RAG baselines (Figure 2)", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-5-efficiency-costs", | |
| "title": "Efficiency measurements show large and inconsistent memory-operation and prediction-time costs for existing memory-based LLM systems (Figure 3)", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "claim-6-feedback-ab", | |
| "title": "Comparisons with and without feedback show simulated user feedback can improve model performance on task-specific metrics (Table 11)", | |
| "children": [] | |
| }, | |
| { | |
| "slug": "conclusion", | |
| "title": "Conclusion", | |
| "children": [] | |
| } | |
| ] | |
| }, | |
| "revision": 1 | |
| } |