diff --git a/.gitattributes b/.gitattributes
index a6344aac8c09253b3b630fb776ae94478aa0275b..51384f8b9938323a2793683cd4096969dc1e0366 100644
--- a/.gitattributes
+++ b/.gitattributes
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
+artifacts/pythia-1.4b-final-layer.ttl filter=lfs diff=lfs merge=lfs -text
diff --git a/ARTIFACT_INDEX.md b/ARTIFACT_INDEX.md
new file mode 100644
index 0000000000000000000000000000000000000000..dd479b481970464c8163770971b3c4702317c769
--- /dev/null
+++ b/ARTIFACT_INDEX.md
@@ -0,0 +1,35 @@
+# Active artifact index
+
+Checksums are SHA-256 over the distributed files. Rejected `.translate`, reader,
+residual, and sequence-controller artifacts are research history and are not part
+of this active Hugging Face pack.
+
+## Runtime artifacts
+
+| Filename | Type | Purpose | Compatibility | SHA-256 |
+|---|---|---|---|---|
+| `canonical-p-v1.router` | Safetensors `.router` | Universal canonical ranking and rejection | `pcm-canonical-p-v1` | `29e0728c12f8806c48998579aedcfa389fa398f909098e162710eb2a5709834e` |
+| `pythia-1.4b-final-layer.ttl` | Safetensors `.ttl` | Semantic Pythia query, value, and gate compatibility | Pythia-1.4B, width 2,048, layer 23 | `72ef68d07ee27c37b90432d34d4be5c2c280ae1bcb08236e37a0e458c054d8d7` |
+| `gemma4-e4b-q8-llama.ltl` | JSON `.ltl` | Direct adaptive lexical output control metadata | Recorded Gemma4 Q8 GGUF, tokenizer bundle, and llama.cpp | `7eee07ed813796dd0d25b04b48aec73507b934479ad8902e1b657653c040781a` |
+| `personality-proof.ppkg` | SQLite `.ppkg` | Durable personality proof package | Personality protocol v1 and canonical P v1 | `faa701fb6150738937fc7b756fb5df226638eedb7599127b0d04d83899ea4bf9` |
+
+The Pythia checkpoint and Gemma GGUF are not distributed. Users must obtain them
+under their upstream licenses and set local paths explicitly.
+
+## Active benchmark evidence
+
+| Filename | Purpose | SHA-256 |
+|---|---|---|
+| `phase-b-factorized-representation.json` | Canonical representation probe | `602bf8b8b1d07b8100b13f9842790f87ae6a5229ddc2d66377130d15413a870f` |
+| `phase-b-split-translator.json` | Pythia TTL, routing, causal, and preservation proof | `4a8cdc0c2bc73a1fc10f974f20d88be4902331a07b92928edb089b852bc1a845` |
+| `phase-b-personality-package.json` | Promotion, durability, growth, and CUDA personality proof | `5d17cbd8cecc8398105d3d53ff05e1b7941dd0732ac3d68a761790b64799bf06` |
+| `ppkg-100k-profile.json` | Integrity-boundary lookup profile | `f5ff5233632bf5bf860fefedeec560e10a3f32729f4b6a767d0b3538d9c33a05` |
+| `active-system-audit.json` | Active architecture audit and scaling | `6eb3e401023b5878b2bbfb9055fd921999b571f6216e370b12440b7d343c0a1d` |
+| `active-system-cuda-attribution.json` | Matched Pythia causal attribution | `ffe0e8f32f9a1a059767607f0c5e0432b6baf2a9917dc56877f794458b1a5e1c` |
+| `gemma4-e4b-q8-causal.json` | Bounded Gemma residual evidence retained as baseline | `a15a862dbc5dde547e7124ce3c3f16d015c15b79730bc60b7e92425e471ecb0f` |
+| `gemma-native-prompt-equivalence.json` | Native browser-message and token equivalence | `f6823b9b5a5fd69ebf75c8aadc8efa23fe0a6eaac7a1eaa81c398a3fdb200e79` |
+| `post-turn-memory-review-acceptance.json` | Natural post-turn review acceptance evidence | `89cb901c1665e9fd1515ef9165580a4bd3c4d9774f23976fdc6f366ef88c6b24` |
+| `debug-actions-profile.json` | Bounded `/state` and `/personality` profile | `3ca4cc8ac002dc30bd6c33d1bdae764bfc3e43534498eecc98a5168069784bd4` |
+| `vram-comparison.json` | Matched P-cache, retained-KV, and combined CUDA memory comparison | `1b1e266c3f6513a5708711f09879a6519ce45abdaea7ba16d04f2510f5c1fc8d` |
+
+The generated evidence manifest is [`assets/EVIDENCE_MANIFEST.json`](assets/EVIDENCE_MANIFEST.json).
diff --git a/CITATION.bib b/CITATION.bib
new file mode 100644
index 0000000000000000000000000000000000000000..862ff9edbe61dc404624793ad76fcc630dba143c
--- /dev/null
+++ b/CITATION.bib
@@ -0,0 +1,15 @@
+@software{planner_cache_2026,
+ author = {{Planner Cache contributors}},
+ title = {Planner Cache: A Portable Bounded Semantic-State Layer for Frozen Language Models},
+ year = {2026},
+ note = {Research software and preprint documentation. Author list, release version, repository URL, and DOI pending}
+}
+
+@article{biderman2023pythia,
+ author = {Biderman, Stella and Schoelkopf, Hailey and Anthony, Quentin Gregory and Bradley, Herbie and O'Brien, Kyle and Hallahan, Eric and Khan, Mohammad Aflah and Purohit, Shivanshu and Prashanth, USVSN Sai and Raff, Edward and Skowron, Aviya and Sutawika, Lintang and van der Wal, Oskar},
+ title = {Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling},
+ journal = {Proceedings of the 40th International Conference on Machine Learning},
+ year = {2023},
+ volume = {202},
+ pages = {2397--2430}
+}
diff --git a/CITATION.cff b/CITATION.cff
new file mode 100644
index 0000000000000000000000000000000000000000..be2fd0993e9f192592dea175a8f550fdf8e232e0
--- /dev/null
+++ b/CITATION.cff
@@ -0,0 +1,13 @@
+cff-version: 1.2.0
+message: "If you use Planner Cache, cite the software and accompanying preprint."
+title: "Planner Cache: A Portable Bounded Semantic-State Layer for Frozen Language Models"
+type: software
+version: 0.1.0
+date-released: 2026-08-23
+authors:
+ - name: "Planner Cache contributors"
+abstract: >-
+ A portable bounded semantic-state layer for frozen language models with a
+ canonical P-cache, universal router, model-specific TTL and LTL compatibility,
+ and disk-resident P-package personality state.
+license: "LicenseRef-Proprietary-NoGrant"
diff --git a/FINAL_RELEASE_AUDIT.md b/FINAL_RELEASE_AUDIT.md
new file mode 100644
index 0000000000000000000000000000000000000000..0ad612089611cf87d8e21b15333fc76659508aaa
--- /dev/null
+++ b/FINAL_RELEASE_AUDIT.md
@@ -0,0 +1,236 @@
+# Planner Cache final release audit
+
+Audit date: 2026-08-23
+
+## Release readiness
+
+The implementation and publication packs are technically validated as a release
+candidate. Public redistribution is **blocked** because the repository does not
+contain a repository-wide software license grant. Final author metadata and a
+public release URL are also missing. No license was invented during this audit.
+
+The publication packs intentionally exclude model weights, GGUF files, tokenizer
+and metadata bundles, datasets, llama.cpp files, and other third-party copyrighted
+payloads. They contain project-authored implementation, documentation, adapters,
+benchmark records, and derived assets only.
+
+## Current architecture
+
+The audited active boundary is:
+
+```text
+Recent KV and runtime history
+ ↓
+frozen model and native chat template
+
+canonical P-cache and selected P-package state
+ ↓
+universal .router
+ ↓
+native P support, semantic .ttl, or lexical .ltl
+```
+
+Recent KV, archive/history, and tool retrieval remain model or runtime
+responsibilities. P-cache is bounded mutable current state. P-package is durable
+disk-resident personality state. The hidden post-turn memory review observes the
+latest exchange as a side-channel and applies only validated canonical P
+operations. It does not rewrite the visible message path.
+
+## Fixed BLOCKER and MAJOR issues
+
+| Severity | Finding | Resolution |
+|---|---|---|
+| BLOCKER | The first staging pass copied the upstream Gemma tokenizer bundle | Removed from every pack. The builder and validator now reject tokenizer, model, GGUF, and common checkpoint payloads |
+| MAJOR | Publication JSON contained workstation-specific absolute paths | Publication copies normalize those paths to portable environment placeholders. Authoritative repository artifacts remain unchanged |
+| MAJOR | Active public exports still exposed rejected residual and lexical research APIs | Removed rejected adapters from the active planner package exports. Historical modules and evidence remain available for research regression |
+| MAJOR | Launch scripts contained machine-specific model and llama.cpp defaults | Replaced model defaults with required environment inputs and made the llama.cpp default home-relative |
+| MAJOR | A clean source checkout could not collect tests without an editable install | Added `src` to the pytest configuration |
+| MAJOR | `/personality` hydrated and serialized the complete package | Added bounded inspection with a default 100-entry page. The 100,000-entry case fell from 9.7874 seconds and 173,110,748 peak Python allocation bytes to 0.0218 seconds and 196,288 bytes for the action |
+| MAJOR | The VRAM comparison initially included first-use CUDA allocations in one condition | Added a matched warm-up. Every recorded row now begins at the same loaded-stack baseline |
+| MAJOR | Publication artifact indexes could diverge after portable path normalization | Pack building now refreshes evidence and artifact checksums after normalization |
+
+## Remaining BLOCKER and MAJOR findings
+
+| Rank | Severity | Finding | Release consequence |
+|---:|---|---|---|
+| 1 | BLOCKER | No repository-wide software license grant exists | Do not publish or redistribute the staged packs until the rights holder adds a license |
+| 2 | BLOCKER | Final authors, affiliations, public repository URL, and release identifier are unset | Citation and preprint metadata remain provisional |
+| 3 | MAJOR | Trained semantic TTL support is proven only for Pythia-1.4B | Do not claim universal or multi-model semantic compatibility |
+| 4 | MAJOR | Natural memory review is narrow and slow | The controlled reviewer targets owner, location, and status. Recorded review latency was 43.14 to 65.50 seconds |
+| 5 | MAJOR | Canonical representation weights are reconstructed rather than shipped as a standalone protocol artifact | Exact third-party reproduction depends on the documented construction path |
+| 6 | MAJOR | Pythia router-index hydration is linear on each wrapper call | Controlled routing accuracy is strong through 1,024 slots, but arbitrary-scale latency is not established |
+
+No other BLOCKER or MAJOR correctness issue was found in the release-focused
+audit. Nuanced personality learning, broader natural-language extraction, large
+debug offsets, multi-seed statistics, and wider model portability remain MINOR,
+OPTIMIZATION, or documented research limitations depending on intended use.
+
+## Component scorecard
+
+| Component | Correctness | Integrity | Performance | Status |
+|---|---|---|---|---|
+| P-cache | Mutation, merge, invalidation, capacity, stale-state, and serialization regressions pass | Canonical snapshots reject corruption and protocol mismatch | Bounded allocation verified | CLEAN |
+| Universal `.router` | Controlled top-1, top-4 recall, and MRR are 1.0 through 1,024 slots | Deterministic checksummed artifact | 1,024-slot measured routing was 0.675 ms. Per-call index hydration remains a MAJOR limitation | CLEAN with documented scaling limitation |
+| Pythia `.ttl` | Relevant P changes causal logits and tested inactive paths reproduce base candidate logits | Model, width, protocol, type, and checksum checks pass | Frozen base has zero gradients. Active cost is included in the matched VRAM run | CLEAN for the proven Pythia configuration |
+| Gemma `.ltl` | Exact routed lexical control is proven for the recorded direct adaptive logit-bias benchmark | Runtime, model, tokenizer checksum, protocol, class, and checksum checks pass | Zero learned parameters. Rejected routes create no lexical target | CLEAN within lexical or output support |
+| `.ppkg` | Promotion, authority, contradiction, context, cold reload, and selective hydration tests pass | Checksum work occurs at integrity boundaries, not normal lookup | 100,000 entries use 152 candidate headers and hydrate four rows in the recorded query | CLEAN for the mechanical proof |
+| Gateway | Inactive P and LTL preserve exact browser messages, rendered prompt, and token IDs | Session files and event logs are structured and deterministic where required | Review is post-response but must finish before the next turn | CLEAN with review-latency limitation |
+
+## Prompt transparency and inert paths
+
+The native Gemma equivalence artifact records identical structured-message,
+rendered-prompt, and token-ID SHA-256 values for the gateway and raw llama-server
+when P and LTL are inactive. The prompt contained 33 tokens. No logit bias was
+present. Wrong-entity, wrong-relation, historical, invalidated, router-disabled,
+and compatibility-disabled paths remain inert in the tested causal regressions.
+
+## Natural memory review
+
+The controlled acceptance run recorded a natural RP CREATE followed by MODIFY:
+
+```text
+brass key.location = kitchen drawer
+brass key.location = coat pocket
+```
+
+The final active state contained only `coat pocket`. The same conceptual review
+path ran for Gemma and Pythia. Unsupported assistant claims and malformed review
+output remain fail-closed in regression tests. The reviewer does not receive or
+alter the visible browser request.
+
+## Exact VRAM comparison
+
+### Command
+
+```bash
+PYTHONPATH=src .venv/bin/python benchmarks/compare_pcache_kv_vram.py \
+ --model pythia-1.4b \
+ --ttl artifacts/pythia-1.4b-final-layer.ttl \
+ --router artifacts/canonical-p-v1.router \
+ --output artifacts/vram-comparison.json \
+ --workloads 64,256,1024 \
+ --generated-tokens 8 \
+ --seed 317
+```
+
+### Matched configuration
+
+- GPU: NVIDIA GeForce RTX 3050 Laptop GPU with 3,950,575,616 bytes
+- Driver: 610.57.04
+- CUDA runtime: 13.0
+- PyTorch: 2.13.0+cu130
+- Transformers: 5.15.1
+- Model: frozen Pythia-1.4B
+- Batch: 1
+- Base precision: float16
+- TTL precision: float32
+- Generation: greedy argmax
+- Generated tokens: 8
+- Baseline method: one warmed loaded stack followed by CUDA synchronization and peak reset
+
+All memory figures below are MiB. `P bytes` is canonical P tensor allocation.
+`KV bytes` is retained model KV tensor storage. CUDA peaks also include transient
+attention, router, TTL, output, and allocator work.
+
+| Prompt and slots | Condition | P bytes | KV bytes | Base alloc | Base reserved | Peak alloc | Peak reserved | Increment alloc | Increment reserved | Runtime |
+|---:|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|
+| 64 | P-cache only | 0.128 | 0.000 | 2717.183 | 2772.000 | 2724.309 | 2776.000 | 7.125 | 4.000 | 0.2555 s |
+| 64 | KV only | 0.000 | 13.312 | 2717.183 | 2772.000 | 2735.575 | 2788.000 | 18.392 | 16.000 | 0.1754 s |
+| 64 | P-cache plus KV | 0.128 | 13.312 | 2717.183 | 2772.000 | 2735.608 | 2788.000 | 18.425 | 16.000 | 0.2045 s |
+| 256 | P-cache only | 0.513 | 0.000 | 2717.183 | 2772.000 | 2744.347 | 2806.000 | 27.164 | 34.000 | 0.7415 s |
+| 256 | KV only | 0.000 | 49.312 | 2717.183 | 2772.000 | 2794.609 | 2852.000 | 77.426 | 80.000 | 0.1937 s |
+| 256 | P-cache plus KV | 0.513 | 49.312 | 2717.183 | 2772.000 | 2794.739 | 2852.000 | 77.556 | 80.000 | 0.4156 s |
+| 1,024 | P-cache only | 2.052 | 0.000 | 2717.183 | 2772.000 | 2820.674 | 2938.000 | 103.491 | 166.000 | 2.7105 s |
+| 1,024 | KV only | 0.000 | 193.312 | 2717.183 | 2772.000 | 3011.449 | 3096.000 | 294.266 | 324.000 | 0.3753 s |
+| 1,024 | P-cache plus KV | 2.052 | 193.312 | 2717.183 | 2772.000 | 3011.966 | 3114.000 | 294.783 | 342.000 | 1.2812 s |
+
+All nine conditions succeeded. OOM events, failures, fallbacks, and estimated
+values were zero. The raw artifact SHA-256 is
+`1b1e266c3f6513a5708711f09879a6519ce45abdaea7ba16d04f2510f5c1fc8d`.
+See the [raw JSON](artifacts/vram-comparison.json),
+[summary](assets/VRAM_COMPARISON.md), [CSV](assets/vram_comparison.csv), and
+[plot](assets/vram_comparison.svg).
+
+The result distinguishes P-cache and KV allocation. It does not imply that
+semantic state and exact token-level KV are interchangeable.
+
+## Exact validation commands and results
+
+```bash
+GEMMA_MODEL=/path/to/tested-gemma.gguf \
+LLAMA_CPP_DIR=/path/to/llama.cpp \
+.venv/bin/python -m pytest -q
+```
+
+The final result was `126 passed in 285.62 seconds` with the exact local Gemma
+runtime enabled. The separate portable no-path run completed with 119 passed and
+seven exact-runtime skips. The focused exact Gemma subset completed with 33
+passed in 216.16 seconds.
+
+```bash
+PYTHONPATH=src python Publishing/assets/generate_assets.py
+PYTHONPATH=src python Publishing/assets/generate_assets.py
+```
+
+The two runs produced byte-identical SVG and normalized PDF hashes. The current
+architecture PDF SHA-256 is
+`19fad644f3a1e3086a845f07850beec07e20a2352cad000b461c21b6802a2519`.
+
+```bash
+.venv/bin/python Publishing/build_release_packs.py
+.venv/bin/python Publishing/validate_release.py
+bash -n run-pythia.sh run-gemma.sh
+.venv/bin/python -m compileall -q src benchmarks Publishing
+git diff --check
+```
+
+The publication validator requires all three manifests to match, all local links
+to resolve, all JSON to parse, shell and Python syntax to pass, no workstation
+absolute paths, and no third-party model or tokenizer payloads.
+
+## Publication folder validation
+
+| Pack | Contents | Independent validation |
+|---|---|---|
+| GitHub | Developer documentation, active source, launchers, tests, benchmarks, active artifacts, historical result evidence, and assets | Passed manifest, link, syntax, JSON, path, and payload checks |
+| Hugging Face | Artifact cards, active compatibility source, active artifacts, benchmark evidence, runtime requirements, and assets | Passed manifest, link, syntax, JSON, path, and payload checks |
+| Research | Manuscript, experiments, ablations, reproducibility map, benchmark scripts, active and negative-result evidence, and assets | Passed manifest, link, syntax, JSON, path, and payload checks |
+
+Upstream models, tokenizers, llama.cpp, datasets, and the historical third-party
+visual specification are referenced as external prerequisites and are not copied.
+
+## Claims safe to publish
+
+- Planner Cache maintains bounded mutable semantic state independently of retained token-level conversation history.
+- The canonical router reached top-1 accuracy and MRR 1.0 through 1,024 slots on the recorded controlled audit.
+- The Pythia TTL provides tested internal causal state compatibility with frozen-base gradient isolation.
+- The Gemma LTL provides tested lexical output compatibility and does not establish internal semantic reasoning.
+- Tested inactive and rejected paths preserve base behavior.
+- P-package provides deterministic checksummed persistence, evidence-based promotion, selective loading, and zero inactive VRAM in the recorded proof.
+- The indexed 100,000-entry P-package query hydrated four entries from 152 candidate headers.
+- The gateway preserves native Gemma messages and tokenization when memory output control is inactive.
+- Natural post-turn review can create and modify controlled owner, location, and status state while failing closed.
+- The recorded matched VRAM matrix completed without failure and keeps P-cache and KV measurements conceptually separate.
+
+## Claims not safe to publish
+
+- Universal model compatibility
+- Trained semantic TTL portability beyond Pythia-1.4B
+- Gemma internal semantic reasoning over P
+- Replacement of arbitrary long context, archives, or historical retrieval
+- Production-ready broad natural-memory extraction
+- Production-ready learned personality behavior
+- Constant-time routing at arbitrary scale
+- Multi-seed statistical generality not present in the artifacts
+
+## Final ranked disposition
+
+1. Add an explicit repository-wide software license before redistribution.
+2. Finalize authors, affiliations, repository URL, and release identifier.
+3. Keep all semantic portability claims scoped to Pythia until a second trained TTL exists.
+4. Present natural memory review as a controlled, narrow, high-latency proof.
+5. Publish a standalone canonical representation weight artifact if exact external reconstruction becomes a release requirement.
+6. Treat per-call router-index hydration as measured technical debt rather than claiming arbitrary-scale routing.
+
+Subject to the two publication metadata blockers, the code, artifacts, evidence,
+and publication packs form a technically clean release candidate.
diff --git a/LICENSE_STATUS.md b/LICENSE_STATUS.md
new file mode 100644
index 0000000000000000000000000000000000000000..1b49984ce1d931caa51e0f1019e0c87b48cd6d65
--- /dev/null
+++ b/LICENSE_STATUS.md
@@ -0,0 +1,16 @@
+# Software license status
+
+No repository-wide software license grant is present in the source repository as
+of 2026-08-23. This publication pack is technically staged but is not authorized
+for public redistribution until the rights holder selects and adds a software
+license.
+
+`THIRD_PARTY_NOTICES.md` records known historical dataset licenses. Base models,
+GGUF files, llama.cpp, Python dependencies, and other third-party components are
+not relicensed by Planner Cache and remain governed by their upstream terms.
+
+Model weights, tokenizer and metadata bundles, datasets, llama.cpp files, and
+other third-party copyrighted payloads are intentionally excluded from every
+publication pack.
+
+This status file is intentionally not a substitute for a license.
diff --git a/LTL_CARD.md b/LTL_CARD.md
new file mode 100644
index 0000000000000000000000000000000000000000..e37d0c93bbf99ebef3e67c8fcccebd4a7734109a
--- /dev/null
+++ b/LTL_CARD.md
@@ -0,0 +1,27 @@
+# Lexical Translation Layer artifact card
+
+A Lexical Translation Layer converts an accepted canonical value into tokenizer or output controls. It can make a runtime emit the routed value without claiming that the model represented or reasoned over that value internally.
+
+## Artifact
+
+`gemma4-e4b-q8-llama.ltl`
+
+| Field | Value |
+|---|---|
+| Adapter class | LTL |
+| Support level | lexical or output |
+| Format | `planner-cache-ltl-v1` |
+| Model | Gemma4 E4B Q8 GGUF |
+| Runtime | llama.cpp |
+| Control | direct adaptive logit bias |
+| Parameters | 0 |
+| Canonical protocol | `pcm-canonical-p-v1` |
+| GGUF SHA-256 | `a4c4177f9fd7e3f56522675afb742f079a53f9226195b7db5e9888c872f053da` |
+| Tokenizer bundle SHA-256 | `b3033e12af0ed503d8b80390c79d02d6bd9bc372e93e377cc1dd6514b7cd21d6` |
+| Artifact SHA-256 | `7eee07ed813796dd0d25b04b48aec73507b934479ad8902e1b657653c040781a` |
+
+The LTL receives only a universal-router accepted canonical value. It tokenizes the exact UTF-8 value and applies one lexical target at a time. Rejected entity, relation, historical, invalidated, router-disabled, and LTL-disabled paths create no target and remain base-equivalent.
+
+The matched 128-value complexity audit recorded 128 of 128 exact strings for direct adaptive logit bias with active-path KL 3.2561. KL divergence measures how far the full output distribution moved from the frozen base. The selected artifact stores identity and control metadata only. It contains no learned tensors, base weights, vocabulary table, P contents, conversation state, or optimizer state.
+
+LTL support proves routed lexical emission. It does not prove internal semantic reasoning over P. Historical Gemma residual and sequence `.translate` artifacts remain research evidence and are not TTLs.
diff --git a/MODEL_CARD.md b/MODEL_CARD.md
new file mode 100644
index 0000000000000000000000000000000000000000..51e95a8361f93f029fc8fa861f1328a78671259e
--- /dev/null
+++ b/MODEL_CARD.md
@@ -0,0 +1,80 @@
+# Planner Cache system card
+
+## Summary
+
+Planner Cache is an external semantic-memory architecture attached to a frozen decoder. In plain terms, it keeps mutable facts in a separate bounded store and supplies only selected facts to the model. This card describes the tested system and does not present Planner Cache as a newly pretrained foundation model.
+
+## Tested base model
+
+| Field | Value |
+|---|---|
+| Base | Pythia-1.4B local checkpoint |
+| Architecture | GPT-NeoX decoder |
+| Layers | 24 |
+| Hidden width | 2048 |
+| Attention heads | 16 |
+| Context positions in bundled config | 2,048 |
+| Base training during Planner Cache experiments | none |
+| Base parameters receiving gradients in active CUDA proof | 0 |
+
+The portability suite also ran a frozen Gemma 4 E4B Q8 GGUF with architecture `gemma4`, 42 layers, hidden width 2,560, and llama.cpp build 10276. A bounded two-value residual proved causal control but not broad compatibility. A later sequence path reached 125 of 128 exact disjoint strings. The matched audit classified that result as lexical token forcing and selected a zero-parameter direct adaptive logit-bias LTL that reached 128 of 128 exact strings. Gemma does not currently have TTL support.
+
+Pythia remains subject to the language, factuality, bias, and safety limitations documented by EleutherAI.
+
+## Added components
+
+- A 512-wide canonical P-cache with fixed configured capacity.
+- A model-independent canonical router.
+- A 2,707,464-parameter Pythia `.ttl` attached at layer 23.
+- A disk-resident `.ppkg` proof with selective canonical activation.
+- A zero-parameter Gemma `.ltl` using direct adaptive lexical control in llama.cpp.
+- A hidden post-turn review side-channel that proposes validated canonical P operations after visible generation.
+
+No LoRA, base-weight modification, prompt prefix, or P state in self-attention KV is used by the active architecture.
+
+## Intended use
+
+- Research on mutable semantic state after source tokens leave recent KV.
+- Evaluation of explicit state creation, modification, merging, invalidation, and retention.
+- Development of small compatibility modules for frozen decoder models.
+- Research on durable evidence-based personality conclusions stored outside model weights.
+
+## Out of scope
+
+- General-purpose long-context replacement.
+- Exact transcript recall without an external archive.
+- Production user profiling.
+- Claims of compatibility with arbitrary decoder models.
+- Safety-critical state tracking without external validation.
+- Foundation-model quality or safety evaluation.
+
+## Evaluation methodology
+
+The active tests use controlled entity, relation, value, mutation, wrong-state, invalidation, held-out composition, natural-RP preservation, persistence, corruption, capacity, and scaling workloads. Source-state tokens are removed from recent KV in causal tests. The same prompt and KV are used while canonical P changes.
+
+The active audit also profiles query construction, index hydration, canonical routing, P-package header filtering, row hydration, canonical conversion, translator latency, generation latency, CPU memory, VRAM, disk size, and bytes loaded.
+
+The gateway transparency regression compares the final tokenized prompt with raw
+llama-server for identical `system` and `user` messages and requires exact token
+equivalence while P and LTL are inactive. Memory review is a separate request.
+
+## Main findings
+
+Changing only canonical P changed the selected answer in the controlled causal tests. Rejecting or invalidating that state restored the frozen output. This shows a causal memory channel in the tested conditions. It does not show broad reasoning, factuality, or universal model support.
+
+- P-cache allocation remains fixed for configured capacity.
+- Relevant canonical P state causally changes Pythia token logits and generated values.
+- Invalidated and rejected wrong-state conditions restore frozen-base logits in the matched CUDA benchmark.
+- Natural-RP preservation remains exact on the held-out proof fixture for irrelevant state.
+- P-package state remains on disk and only selected entries are activated.
+- The Gemma LTL audit emitted all 128 selected held-out strings. Rejected, invalidated, and disabled paths remained inert. This is lexical compatibility, not evidence of internal semantic reasoning.
+- A matched CUDA matrix distinguishes canonical P allocation, retained KV tensors, and combined peak VRAM at 64, 256, and 1,024 tokens or slots without treating P and KV as interchangeable.
+
+## Known limitations
+
+The evaluation fixtures are synthetic or small held-out RP sets. Natural review
+currently covers a narrow state schema and is slow. The results do not establish
+general factuality, broad instruction following, production dialogue quality,
+nuanced personality, or a full trained open-vocabulary TTL on a second model
+family. Gemma LTL exact-string performance must not be presented as learned
+semantic compatibility.
diff --git a/PPKG_CARD.md b/PPKG_CARD.md
new file mode 100644
index 0000000000000000000000000000000000000000..06d0457038ad387447ea1bff535a7c93b83b1d71
--- /dev/null
+++ b/PPKG_CARD.md
@@ -0,0 +1,69 @@
+# P-package `.ppkg` card
+
+## Artifact
+
+`personality-proof.ppkg`
+
+## Format
+
+| Field | Value |
+|---|---|
+| Container | SQLite |
+| Format | `pcm-personality-package-v1` |
+| Protocol | `pcm-canonical-personality-v1` |
+| Canonical compatibility | `pcm-canonical-p-v1` |
+| File SHA-256 | `faa701fb6150738937fc7b756fb5df226638eedb7599127b0d04d83899ea4bf9` |
+
+## Purpose
+
+P-package, stored as `.ppkg`, keeps durable personality and behavioral conclusions on disk. It promotes a conclusion only after repeated, diverse, or authoritative evidence. It is not a transcript archive and does not store model-native vectors.
+
+## Included logical data
+
+- canonical personality entries
+- evidence candidates and archive references
+- support and contradiction IDs
+- source authority
+- scope and relationship identity
+- strength, confidence, and importance
+- timestamps and status
+- reversible before and after changes
+- protocol metadata and semantic checksum
+
+## Excluded data
+
+- raw conversation transcript
+- base-model weights
+- model token IDs
+- hidden vectors
+- `.ttl` or `.ltl` contents
+- LoRA
+- optimizer state
+
+## Mechanical results
+
+The practical result is conservative promotion. One weak event did nothing, repeated independent evidence promoted, cross-context evidence counted more than narrow repetition, and explicit correction overruled unsupported model claims. This tests auditable memory mechanics, not nuanced personality understanding.
+
+One weak event did not promote. Three independent events promoted. Linked cross-context evidence scored 2.5865 compared with 1.8000 for five narrow-context events. Unsupported model claims did not promote. An explicit correction promoted and superseded the old conclusion while retaining audit history.
+
+Technical, creative, and relationship-specific retrieval passed in the proof workload. Irrelevant retrieval loaded zero entries.
+
+## Causal results
+
+The same prompt and KV generated Alice when the selected package conclusion specified Alice and Bob when it specified Bob. Irrelevant and unpromoted low-confidence packages reproduced frozen-base logits. The proof uses controlled preferred-persona values inside the tested compatibility range.
+
+## Scaling
+
+The indexed audit measured a 58,720,256-byte synthetic package at 100k entries. Header routing took 67.10 ms, selected-row hydration took 0.118 ms, canonical conversion took 0.495 ms, and four entries were loaded from 152 candidate headers. Inactive VRAM was zero.
+
+The separate `/personality` inspection view is paginated. At 100k active entries,
+its default 100-entry page measured 21.8 ms before JSON encoding and produced a
+roughly 60 KB response instead of hydrating the entire package.
+
+## Integrity behavior
+
+Full semantic verification occurs on verified open, explicit verify, export, checkpoint, and dirty close. Normal routing and evidence pushes do not hash the complete package. SQLite transactional integrity remains enabled.
+
+## Limitations
+
+The package proof is mechanical. It does not establish nuanced learned personality. Canonical conversion depends on an external factorized representation recipe that is not yet distributed as a separate checksummed artifact.
diff --git a/README.md b/README.md
index 154df8298fab5ecf322016157858e08cd1bccbe1..0dbeacbddca802f4506037e87f32ecac932cabd2 100644
--- a/README.md
+++ b/README.md
@@ -1,3 +1,133 @@
---
-license: apache-2.0
+tags:
+- semantic-memory
+- pythia
+- pytorch
+- safetensors
+- sqlite
+library_name: transformers
---
+
+# Planner Cache artifacts
+
+Planner Cache is an external memory layer for frozen language models. It stores a bounded set of current facts outside the prompt, selects relevant facts, and exposes them through a small model or runtime compatibility artifact. This repository contains those Planner Cache artifacts. It is not a foundation model.
+
+Planner Cache is not a foundation model and does not replace arbitrary long context. Recent KV, exact history, and tool retrieval remain responsibilities of the model runtime.
+
+## Terms
+
+- **P-cache** is a fixed-capacity store for facts that are currently true.
+- **Canonical P** is the model-independent structured representation of those facts. It contains no model token IDs or hidden vectors.
+- **Router or `.router`** selects the canonical state relevant to the current entity and relation.
+- **TTL or `.ttl`** means Tensor Translation Layer. It converts canonical P into a model's internal state and represents semantic or internal support.
+- **LTL or `.ltl`** means Lexical Translation Layer. It converts a routed value into tokenizer or output controls and represents lexical or output support.
+- **P-package or `.ppkg`** stores durable personality patterns on disk and loads only selected entries.
+- **Retained KV** is recent token-level attention memory maintained by the model runtime.
+- An **active path** accepted memory and enabled TTL or LTL. An **inactive path** rejected, invalidated, or disabled memory and should match the frozen base.
+- A **causal intervention** keeps the prompt fixed and changes only P. **KL divergence** measures how much the output distribution changed. **Incremental VRAM** is extra peak GPU memory above a warmed baseline.
+
+
+
+## Distributed artifacts
+
+| Artifact | Role | Compatibility |
+|---|---|---|
+| `canonical-p-v1.router` | Universal canonical state ranking and rejection | `pcm-canonical-p-v1` |
+| `pythia-1.4b-final-layer.ttl` | Semantic query, value, gate, and layer metadata | Pythia-1.4B, hidden width 2048, layer 23 |
+| `gemma4-e4b-q8-llama.ltl` | Direct adaptive lexical control metadata | Recorded Gemma4 Q8 GGUF and tokenizer checksums, llama.cpp |
+| `personality-proof.ppkg` | Example durable personality package | `pcm-canonical-personality-v1` and canonical P v1 |
+| Benchmark JSON | Raw recorded results | See [ARTIFACT_INDEX.md](ARTIFACT_INDEX.md) |
+
+## Compatibility boundary
+
+```text
+canonical P state
+ -> universal .router
+ -> selected canonical value
+ -> native P, model-specific .ttl, or runtime-specific .ltl
+ -> frozen decoder
+```
+
+The canonical router has no model hidden dimension. A `.ttl` contains semantic compatibility weights. An `.ltl` contains lexical control metadata and may have no learned parameters. Neither contains base weights or P contents. The `.ppkg` contains canonical personality entries and evidence references but no model-native tensors.
+
+The Gemma GGUF, tokenizer files, llama.cpp binaries, and upstream model metadata
+are not redistributed. Users must supply the exact compatible bundle identified
+by the LTL checksums.
+
+## Loading the router and TTL
+
+```python
+from transformers import AutoModelForCausalLM
+
+from pcm.planner import ByteEntityEncoder, CanonicalPRouter
+from pcm.planner import PythiaSplitTranslatedModel, TensorTranslationLayer
+
+base = AutoModelForCausalLM.from_pretrained(
+ "EleutherAI/pythia-1.4b",
+ dtype="float16",
+).to("cuda")
+
+ttl = TensorTranslationLayer.load(
+ "pythia-1.4b-final-layer.ttl",
+ device="cuda",
+)
+router = CanonicalPRouter.load("canonical-p-v1.router", device="cuda")
+model = PythiaSplitTranslatedModel(base, ttl, router, ByteEntityEncoder())
+```
+
+The public Hub repository will need to provide the Planner Cache Python implementation or a pinned source release. The artifacts are not standalone Transformers models.
+
+## Loading P-package
+
+```python
+from pcm.planner import PersonalityPackage, PersonalityQuery, PersonalityRouter
+
+with PersonalityPackage("personality-proof.ppkg") as package:
+ selection = PersonalityRouter().retrieve(
+ package,
+ PersonalityQuery(
+ subject="user",
+ interaction_type="technical",
+ domain="debugging",
+ relation="response_style",
+ ),
+ top_k=4,
+ )
+```
+
+## Measured results
+
+The central causal result is that changing only valid P state changed the tested answer, while wrong, historical, invalidated, or disabled state left the tested base logits unchanged. At the 1,024-unit memory case, canonical P occupied about 2.05 MiB and retained KV tensors occupied about 193.31 MiB, a roughly 94-fold representation-size difference. These stores have different purposes and are not interchangeable.
+
+- Post-audit canonical routing reached 100% top-1 and MRR 1.0 through 1,024 slots on the recorded synthetic scaling workload.
+- The selected Pythia TTL reached 100% controlled held-out state generation at the 128-slot proof target.
+- Wrong entity, wrong relation, historical, invalidated, router-disabled, and TTL-disabled matched CUDA conditions restored frozen-base logits where expected.
+- The frozen Pythia base had zero parameters receiving gradients.
+- The Gemma Q8 LTL emitted 128 of 128 held-out selected strings through direct adaptive logit bias. Inactive paths were exact.
+- The Gemma LTL has zero learned parameters, occupies 708 bytes in the local artifact, adds no prompt tokens, and uses zero inactive VRAM.
+- Indexed `.ppkg` header routing measured 67.10 ms at 100k entries and hydrated four rows from 152 headers.
+- Opening an inactive `.ppkg` changed CUDA allocation by zero bytes.
+- Natural post-turn review created and modified current RP state without explicit memory syntax in both interactive paths. It remains a controlled, slow extraction proof.
+- The matched 1,024-token CUDA workload measured 103.491 MiB incremental peak for P-only with retained KV disabled, 294.266 MiB for retained KV only, and 294.783 MiB with both active.
+
+See [TTL_CARD.md](TTL_CARD.md), [LTL_CARD.md](LTL_CARD.md), [ROUTER_CARD.md](ROUTER_CARD.md), [PPKG_CARD.md](PPKG_CARD.md), and the raw evidence in [ARTIFACT_INDEX.md](ARTIFACT_INDEX.md).
+
+## Limitations
+
+- Full trained semantic compatibility is proven only with Pythia-1.4B.
+- The GPT-2 proof is structural and uses a tiny random model.
+- Gemma has LTL support, not TTL support. Exact lexical emission does not establish internal semantic reasoning over P.
+- A separate sequence-aware prototype reached 125 of 128 exact disjoint strings. It is not the selected runtime artifact and has not passed the 1,000-value or broad active-RP gates.
+- The sequence complexity audit showed that exact performance comes primarily from tokenizer IDs and per-token forcing. Direct logit bias reached 128 of 128 with lower KL, but it is a lexical constraint rather than semantic translation.
+- Canonical representation weights are reconstructed from a fixed recipe instead of being shipped as a standalone versioned artifact.
+- Router-index hydration is linear in configured slot count.
+- P-package personality behavior is a controlled deterministic proof, not nuanced neural personality learning.
+- Planner Cache does not preserve exact old wording and does not replace archive retrieval.
+- The reviewer currently focuses on owner, location, and status state. Review failures are inert and may miss valid facts. Recorded review latency was tens of seconds on the test hardware.
+
+## Licensing and release metadata
+
+Pythia-1.4B is Apache-2.0 according to its model card. Historical dataset
+licenses are included in `THIRD_PARTY_NOTICES.md`. This project currently has no
+repository-wide software license grant. Public redistribution remains blocked
+until the rights holder selects one.
diff --git a/RELEASE_MANIFEST.json b/RELEASE_MANIFEST.json
new file mode 100644
index 0000000000000000000000000000000000000000..6b6e5bebffe77eb812081e99d148ce6a3b874be6
--- /dev/null
+++ b/RELEASE_MANIFEST.json
@@ -0,0 +1,249 @@
+{
+ "files": {
+ "ARTIFACT_INDEX.md": {
+ "bytes": 3153,
+ "sha256": "4317c3e5c4a1d5a1729a4fd43454b1ada2baa5e136c3e110a3b6a89e6cb31b98"
+ },
+ "CITATION.bib": {
+ "bytes": 867,
+ "sha256": "29b4ec696ec4a643740decdccde41befa8df470c378032b3f5442667ad5938d6"
+ },
+ "CITATION.cff": {
+ "bytes": 557,
+ "sha256": "e0ae2c1741b91795156b26f05999a0317c04415a01ef8ecf0400b0c3ab82fb11"
+ },
+ "FINAL_RELEASE_AUDIT.md": {
+ "bytes": 14166,
+ "sha256": "52df283843415aaf6dfa324ca472724584326a8857b7b1b283d16c1a182e6048"
+ },
+ "LICENSE_STATUS.md": {
+ "bytes": 757,
+ "sha256": "561e1cb080eaab253187388e5be2dc9276a9c5a3a8852dbb4a424fdf48c783d2"
+ },
+ "LTL_CARD.md": {
+ "bytes": 1761,
+ "sha256": "32eee111983eb6f5c9b38933d7adfaa6c4a4ec33ee9cd297d4a753f912c6fd02"
+ },
+ "MODEL_CARD.md": {
+ "bytes": 4950,
+ "sha256": "ab7e499220b6f5a70de79ff71d5de8d1bb99bcbe608c4e567a528aee6473d33a"
+ },
+ "PPKG_CARD.md": {
+ "bytes": 3181,
+ "sha256": "b60c082fc5facdd7b7599fa635bb99e4404d48dd9a599bda0f0465714e199683"
+ },
+ "README.md": {
+ "bytes": 7877,
+ "sha256": "ffda2a0324dc5c2692a3afac2b59b183a07246c84deeda0253f87fc7e4e85804"
+ },
+ "ROUTER_CARD.md": {
+ "bytes": 2737,
+ "sha256": "76762ee2d80c768169b452ca1d27724efa5167fea31a60ee0550680c48d09120"
+ },
+ "RUNTIME.md": {
+ "bytes": 1406,
+ "sha256": "bf5900e8fef5ffc982c4869ee0e20ff06c654063f974a8036f32ad4b9f0ae31d"
+ },
+ "SANITIZATION_REPORT.md": {
+ "bytes": 3411,
+ "sha256": "c7a609f4d464772dce34c3eb0ca79e2fda03b40bb75f518b53124973c00b677e"
+ },
+ "THIRD_PARTY_NOTICES.md": {
+ "bytes": 208,
+ "sha256": "c0b88b4f2f86d9f80d0c64bb0a73a1d29ba00f29b85966f9b5d144db244ecd47"
+ },
+ "TTL_CARD.md": {
+ "bytes": 1619,
+ "sha256": "a16b2432000b6096c215a5f00b9ff4f0d11aa28c3030fff2338158aa827019ad"
+ },
+ "artifacts/active-system-audit.json": {
+ "bytes": 32951,
+ "sha256": "6eb3e401023b5878b2bbfb9055fd921999b571f6216e370b12440b7d343c0a1d"
+ },
+ "artifacts/active-system-cuda-attribution.json": {
+ "bytes": 11695,
+ "sha256": "ffe0e8f32f9a1a059767607f0c5e0432b6baf2a9917dc56877f794458b1a5e1c"
+ },
+ "artifacts/canonical-p-v1.router": {
+ "bytes": 592,
+ "sha256": "29e0728c12f8806c48998579aedcfa389fa398f909098e162710eb2a5709834e"
+ },
+ "artifacts/debug-actions-profile.json": {
+ "bytes": 7528,
+ "sha256": "3ca4cc8ac002dc30bd6c33d1bdae764bfc3e43534498eecc98a5168069784bd4"
+ },
+ "artifacts/gemma-native-prompt-equivalence.json": {
+ "bytes": 1190,
+ "sha256": "f6823b9b5a5fd69ebf75c8aadc8efa23fe0a6eaac7a1eaa81c398a3fdb200e79"
+ },
+ "artifacts/gemma4-e4b-q8-causal.json": {
+ "bytes": 13550,
+ "sha256": "a15a862dbc5dde547e7124ce3c3f16d015c15b79730bc60b7e92425e471ecb0f"
+ },
+ "artifacts/gemma4-e4b-q8-llama.ltl": {
+ "bytes": 708,
+ "sha256": "7eee07ed813796dd0d25b04b48aec73507b934479ad8902e1b657653c040781a"
+ },
+ "artifacts/personality-proof.ppkg": {
+ "bytes": 61440,
+ "sha256": "faa701fb6150738937fc7b756fb5df226638eedb7599127b0d04d83899ea4bf9"
+ },
+ "artifacts/phase-b-factorized-representation.json": {
+ "bytes": 1518,
+ "sha256": "602bf8b8b1d07b8100b13f9842790f87ae6a5229ddc2d66377130d15413a870f"
+ },
+ "artifacts/phase-b-personality-package.json": {
+ "bytes": 11471,
+ "sha256": "5d17cbd8cecc8398105d3d53ff05e1b7941dd0732ac3d68a761790b64799bf06"
+ },
+ "artifacts/phase-b-split-translator.json": {
+ "bytes": 30619,
+ "sha256": "4a8cdc0c2bc73a1fc10f974f20d88be4902331a07b92928edb089b852bc1a845"
+ },
+ "artifacts/post-turn-memory-review-acceptance.json": {
+ "bytes": 1814,
+ "sha256": "89cb901c1665e9fd1515ef9165580a4bd3c4d9774f23976fdc6f366ef88c6b24"
+ },
+ "artifacts/ppkg-100k-profile.json": {
+ "bytes": 1890,
+ "sha256": "f5ff5233632bf5bf860fefedeec560e10a3f32729f4b6a767d0b3538d9c33a05"
+ },
+ "artifacts/pythia-1.4b-final-layer.ttl": {
+ "bytes": 10832776,
+ "sha256": "72ef68d07ee27c37b90432d34d4be5c2c280ae1bcb08236e37a0e458c054d8d7"
+ },
+ "artifacts/vram-comparison.json": {
+ "bytes": 9347,
+ "sha256": "1b1e266c3f6513a5708711f09879a6519ce45abdaea7ba16d04f2510f5c1fc8d"
+ },
+ "assets/EVIDENCE_MANIFEST.json": {
+ "bytes": 2267,
+ "sha256": "90a7ea25fafcbdb63fb16d5843cc4ea10b6af30da6fc88aa3348546e11424c94"
+ },
+ "assets/README.md": {
+ "bytes": 2326,
+ "sha256": "5b50dcb78bc07e0b81e38680a77b299c4a791264f914d9e55e1c7235ddfac66a"
+ },
+ "assets/VRAM_COMPARISON.md": {
+ "bytes": 3116,
+ "sha256": "515fa0b0ed924670010d7c522fd6b07eef62954645c5ef2c6bcfe2f510569164"
+ },
+ "assets/architecture.mmd": {
+ "bytes": 536,
+ "sha256": "a3501af51b8381e5500b092a6cea2cc6fc84a0d604fc31360b82cfa2cba20e82"
+ },
+ "assets/architecture.pdf": {
+ "bytes": 33505,
+ "sha256": "19fad644f3a1e3086a845f07850beec07e20a2352cad000b461c21b6802a2519"
+ },
+ "assets/architecture.svg": {
+ "bytes": 5796,
+ "sha256": "896fbe844c3a15bef586f1d0eb1b249560cb482ba3a2891c7fe0dd03b8056156"
+ },
+ "assets/causal_conditions.csv": {
+ "bytes": 2015,
+ "sha256": "b756dcf96c276885d58784df4e25a9de0d1f114e2631e68ba42bd6f2fdb97b80"
+ },
+ "assets/gemma_causal_conditions.csv": {
+ "bytes": 1108,
+ "sha256": "86e490aedf0ade7ef7c5d153887a71fd973fd2c2ec50bed8fd45880891a711a9"
+ },
+ "assets/generate_assets.py": {
+ "bytes": 19721,
+ "sha256": "da2d2b9d710f1bb238445af967065ee06866643957e9b829945b9ee55b7b4f6c"
+ },
+ "assets/ppkg_lookup.svg": {
+ "bytes": 3462,
+ "sha256": "c3cb8293f032d2c893f2f2f308904d2132abeed66bd109713a29e0fcf530bf91"
+ },
+ "assets/ppkg_scaling.csv": {
+ "bytes": 655,
+ "sha256": "3a6998adaed8f7ae59bd7f9beba36a402750d609be7ed9df441678ad87ef0cbe"
+ },
+ "assets/router_scaling.csv": {
+ "bytes": 300,
+ "sha256": "18bae82acb8dc34f40b6600ca1965bc15ddfe8c33546e2579850a03aa9f0f52a"
+ },
+ "assets/router_scaling.svg": {
+ "bytes": 4751,
+ "sha256": "a85d53fb6116baf46cc3bf68507e7073ed2b6e9f62872fb5179464b000246cb0"
+ },
+ "assets/vram_comparison.csv": {
+ "bytes": 1214,
+ "sha256": "e9c105610f4ecfca65a59bd2e271468039e86693d8643bdb155353339ad6c16f"
+ },
+ "assets/vram_comparison.svg": {
+ "bytes": 4163,
+ "sha256": "f27c0503fe378fdb0ad400ddb4280720425167369a2a8c6f4bc990015787cdbb"
+ },
+ "pyproject.toml": {
+ "bytes": 693,
+ "sha256": "b716b42cf7f61328805fde14ea6249c7f9d475de6127532c7bf35ad13c61fa8b"
+ },
+ "src/pcm/__init__.py": {
+ "bytes": 70,
+ "sha256": "13fa02743f631c4d76dc7d4e0b849f698a7ea3c357d0575e2aac49145c3a1e1b"
+ },
+ "src/pcm/planner/__init__.py": {
+ "bytes": 2335,
+ "sha256": "b42895436ecb899f65443d4bc6d9804477b5b7be30354e37a97f2eb0ef05b2fb"
+ },
+ "src/pcm/planner/cache.py": {
+ "bytes": 13974,
+ "sha256": "211cf9ad7baddf07e7e9e3fdbccfa7c04d446105ca014114e823cf67a490cb1f"
+ },
+ "src/pcm/planner/canonical.py": {
+ "bytes": 8251,
+ "sha256": "85f9107db2e7652989114d09b6b16af59deca249d6e632ed9e2c77c01e2eef3f"
+ },
+ "src/pcm/planner/chat_cli.py": {
+ "bytes": 21671,
+ "sha256": "31e4a30b0eb56ab78cc822ead4c79688800a5a5c79558ba0e7f06de0132b773f"
+ },
+ "src/pcm/planner/compatibility.py": {
+ "bytes": 12453,
+ "sha256": "6bfd8d490ccdd9049a4f521082ab1d35e2990cce0aa2f23aa5b38e26948cd368"
+ },
+ "src/pcm/planner/interactive_runtimes.py": {
+ "bytes": 37688,
+ "sha256": "1ab389056a07534a287d96fbff1a4dbd5b8e0f0366e920b8e61c84e2e4bc65d0"
+ },
+ "src/pcm/planner/interactive_session.py": {
+ "bytes": 31970,
+ "sha256": "e45044703b472d0cf61a790d169eee52b326466545758ba92b2bf007fe7734f9"
+ },
+ "src/pcm/planner/memory_review.py": {
+ "bytes": 15500,
+ "sha256": "24b5e98475eeb8190c157215d9e7dd763a7b6ffd70c801cfddd8f3d5975261d8"
+ },
+ "src/pcm/planner/personality.py": {
+ "bytes": 47537,
+ "sha256": "c480148d3134c46ebdef142cc95759e18b89441f43fad08135304e7dda437bdf"
+ },
+ "src/pcm/planner/personality_eval.py": {
+ "bytes": 28514,
+ "sha256": "0f26540826a267b669f9159b020d04045ea74bba4faf283f468c3cf6b46cefab"
+ },
+ "src/pcm/planner/pythia_split_translate.py": {
+ "bytes": 8018,
+ "sha256": "6f421cc12f9f86d965063fa4604317fbade3d025b664145639c14fcf2fee0fb9"
+ },
+ "src/pcm/planner/representation.py": {
+ "bytes": 9797,
+ "sha256": "0a44e7c802dc7aa93bbdb58a64fa8d9719725a92d500b44240d49053057aba60"
+ },
+ "src/pcm/planner/split_translator.py": {
+ "bytes": 16193,
+ "sha256": "8beeb52682c043a02bdafeeb012c95ec01283dbd029b882f0b974a27e9901931"
+ },
+ "src/pcm/planner/split_translator_eval.py": {
+ "bytes": 44711,
+ "sha256": "2e1b301268f806b4d7d08f5ff33f0707392839913dbde85a258bc20ac5f07b79"
+ },
+ "src/pcm/planner/web_chat.py": {
+ "bytes": 14555,
+ "sha256": "3d8ba8557212f339d319eb9ee0410b533ca86806984b85b264f089b3b625e4ad"
+ }
+ },
+ "format": "planner-cache-release-pack-v1"
+}
diff --git a/ROUTER_CARD.md b/ROUTER_CARD.md
new file mode 100644
index 0000000000000000000000000000000000000000..bb29151166fda1145975501372e5b70a625d05ed
--- /dev/null
+++ b/ROUTER_CARD.md
@@ -0,0 +1,54 @@
+# Canonical router card
+
+The canonical router is the model-independent selector. Given a structured query, it ranks current P entries and rejects state that belongs to the wrong entity, relation, time, or validity status.
+
+## Artifact
+
+`canonical-p-v1.router`
+
+## Metadata
+
+| Field | Value |
+|---|---|
+| Format | `pcm-canonical-router-v1` |
+| Architecture | `canonical_factor_router_v1` |
+| Canonical protocol | `pcm-canonical-p-v1` |
+| Entity width | 128 |
+| Relation count | 3 |
+| Metadata count | 4 |
+| Model hidden dimensions | 0 |
+| Learned scalar parameters | 5 |
+| File SHA-256 | `29e0728c12f8806c48998579aedcfa389fa398f909098e162710eb2a5709834e` |
+| Tensor SHA-256 | `5434b21de18a0f66cd4506c057989d18796fad84c55725e33e04688fbc7650bc` |
+
+## Routing semantics
+
+The router consumes a canonical query and a canonical slot index. It scores tokenizer-independent entity similarity, relation agreement, metadata agreement, and current-state status. Invalidated and stale slots are masked. A calibrated acceptance threshold rejects wrong-entity, wrong-relation, historical, invalidated, and irrelevant candidates before translation.
+
+The router does not consume Pythia hidden states. Model hidden states are converted to canonical query fields before routing.
+
+The same router and canonical byte-derived query were used unchanged in the bounded Gemma4 Q8 llama.cpp proof. That proof supplied the canonical query externally because the public runtime path did not expose Gemma hidden states. It therefore validates router reuse but not a Gemma query projector.
+
+## Post-audit scaling
+
+The router chose the correct entry first on every controlled query through 1,024 slots. This fixed the earlier decline without hiding errors by increasing top-k. The result does not remove the separate linear index-hydration cost.
+
+| Slots | Top-1 | Top-4 recall | MRR |
+|---:|---:|---:|---:|
+| 4 | 100% | 100% | 1.0 |
+| 20 | 100% | 100% | 1.0 |
+| 64 | 100% | 100% | 1.0 |
+| 128 | 100% | 100% | 1.0 |
+| 256 | 100% | 100% | 1.0 |
+| 512 | 100% | 100% | 1.0 |
+| 1,024 | 100% | 100% | 1.0 |
+
+These are post-audit router measurements after canonical merge was constrained by entity and relation identity. The immutable pre-fix phase artifact remains 100% at 128, 95% at 256, and 85% at 512.
+
+## Performance
+
+Rank latency remained below 0.7 ms through 1,024 slots in the audit. Index hydration is separate and measured 30.12 ms at 128 slots and 233.00 ms at 1,024 slots.
+
+## Limitations
+
+The current implementation rebuilds byte-derived entity anchors for every slot during each wrapper call. The router uses a linear score over configured slots. The scaling benchmark is controlled and does not establish universal entity disambiguation in open-domain text.
diff --git a/RUNTIME.md b/RUNTIME.md
new file mode 100644
index 0000000000000000000000000000000000000000..2195873c7979fb9c8ce65910f057ba356990026f
--- /dev/null
+++ b/RUNTIME.md
@@ -0,0 +1,33 @@
+# Runtime instructions
+
+Planner Cache does not distribute base-model weights or llama.cpp. Create the
+Python environment, provide local model paths, and use the included launchers.
+The exact Gemma tokenizer and metadata bundle is also an external upstream
+requirement. Its contents are not redistributed in these publication packs.
+
+```bash
+python -m venv .venv
+.venv/bin/pip install -e '.[dev,publishing]'
+
+export LLAMA_CPP_DIR=/path/to/llama.cpp
+export GEMMA_MODEL=/path/to/compatible-gemma.gguf
+export GEMMA_TOKENIZER_BUNDLE=/path/to/matching-gemma-tokenizer-bundle
+./run-gemma.sh
+
+export PYTHIA_MODEL=/path/to/pythia-1.4b
+export REVIEW_MODEL="$GEMMA_MODEL"
+./run-pythia.sh
+```
+
+The launchers resolve the project root from their own location. Optional paths
+include `PYTHIA_TTL`, `GEMMA_LTL`, `GEMMA_TOKENIZER_BUNDLE`, `ROUTER_PATH`,
+`PPKG_PATH`, `SESSION_ROOT`, `LLAMA_WEB_UI`, `WEB_HOST`, and `WEB_PORT`.
+
+Gemma uses llama.cpp and the active `.ltl`. Pythia uses the semantic `.ttl` and
+uses the configured frozen GGUF as a CPU structured reviewer by default. The
+review request is separate from visible generation and uses neither TTL nor LTL.
+
+The browser is the primary conversation interface. Terminal commands `/state`,
+`/personality`, `/events`, `/save`, and `/quit` are secondary diagnostics.
+Every session records transcript, events, metadata, P-cache state, and final
+state under `sessions/`.
diff --git a/SANITIZATION_REPORT.md b/SANITIZATION_REPORT.md
new file mode 100644
index 0000000000000000000000000000000000000000..f2b2827aea298776abceb342b3cf868ffbdea4db
--- /dev/null
+++ b/SANITIZATION_REPORT.md
@@ -0,0 +1,79 @@
+# Publication sanitization report
+
+Sanitization date: 2026-08-23
+
+## Scope
+
+The GitHub, Hugging Face, and Research packs were rebuilt from the publication
+source set. This pass changed packaging and explanation only. It did not change
+the architecture, run new benchmarks, or alter recorded benchmark values.
+
+## Removed
+
+- Nested `.git` repositories, including local commit identity and email data
+- Cache directories and temporary build output
+- Third-party model weights, GGUF files, tokenizer and configuration bundles,
+ datasets, llama.cpp files, and ordinary checkpoint formats
+- Random temporary-directory identifiers from publication copies of benchmark JSON
+- Concrete interactive session IDs from publication copies of acceptance evidence
+- Workstation-specific repository, model, and runtime paths
+- Machine-specific launcher defaults for local model, tokenizer, and llama.cpp paths
+
+## Replaced
+
+- Repository paths became `${REPOSITORY_ROOT}` where provenance required a path
+- Model paths became `${GEMMA_MODEL}` or neutral `/path/to/model.gguf` examples
+- Runtime paths became `${LLAMA_CPP_DIR}` or `/path/to/llama.cpp`
+- Temporary run directories became `${TEMP_DIR}/planner-cache-run`
+- Concrete session IDs became `benchmark-session-gemma` or
+ `benchmark-session-pythia`
+- Any detected email address in generated pack text becomes `user@example.com`
+
+Path and identifier normalization changes only non-numerical provenance fields in
+the publication copies. The authoritative repository benchmark artifacts remain
+unchanged. Pack-specific checksums are regenerated after normalization.
+
+## Readability changes
+
+The main README, benchmark guide, Hugging Face cards, research abstract, paper,
+and VRAM guide now state the practical result and its boundary before detailed
+tables. P-cache, canonical P, router, TTL, LTL, P-package, retained KV, active
+and inactive paths, causal intervention, KL divergence, and incremental VRAM are
+defined in plain English at first use in each primary publication entry point.
+
+## Intentionally retained technical metadata
+
+The following fields are useful for reproduction and are not treated as personal
+identifiers:
+
+- Model family and architecture identifiers
+- Model, adapter, router, and evidence checksums
+- llama.cpp build number and commit identifier
+- GPU model, VRAM capacity, driver, CUDA, PyTorch, Transformers, Python, kernel,
+ and platform versions
+- Benchmark names, seeds, layer numbers, dimensions, token counts, timestamps,
+ durations, and measured values
+- Synthetic test entities, names, state values, and controlled role-play examples
+
+No hostname, account name, private email, personal conversation, or original
+session identifier is required for reproduction.
+
+## Validation
+
+The release validator checks every pack for nested repository metadata, cache
+directories, private email addresses, personal machine identifiers, concrete
+session IDs, random temporary paths, absolute home or removable-media paths,
+forbidden third-party payloads, broken links, malformed JSON, syntax errors, and
+manifest mismatch.
+
+Validation commands:
+
+```bash
+.venv/bin/python Publishing/build_release_packs.py
+.venv/bin/python Publishing/validate_release.py
+PYTHONPATH=src python Publishing/assets/generate_assets.py
+git diff --check
+```
+
+All publication sanitization, payload, manifest, link, JSON, syntax, asset, and
+diff checks passed in the final run.
diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md
new file mode 100644
index 0000000000000000000000000000000000000000..64f0ce7a4ebdc8e2039b40165749343421a5c09a
--- /dev/null
+++ b/THIRD_PARTY_NOTICES.md
@@ -0,0 +1,6 @@
+## Training data:
+- PIPPA — PygmalionAI — Apache-2.0
+- SPB-2602 — marcoDSN — CC BY 4.0
+- SOC-2508 — marcoDSN — CC BY 4.0
+
+The datasets were normalized, filtered, and deduplicated before training.
diff --git a/TTL_CARD.md b/TTL_CARD.md
new file mode 100644
index 0000000000000000000000000000000000000000..c807034367eb19d7bd5d93cca2ecddcc2028903f
--- /dev/null
+++ b/TTL_CARD.md
@@ -0,0 +1,27 @@
+# Tensor Translation Layer artifact card
+
+A Tensor Translation Layer converts model-independent canonical P into a frozen model's internal hidden-state space. It is intended for semantic or internal memory use rather than direct token forcing.
+
+## Artifact
+
+`pythia-1.4b-final-layer.ttl`
+
+| Field | Value |
+|---|---|
+| Adapter class | TTL |
+| Support level | semantic or internal |
+| Format | `planner-cache-ttl-v1` |
+| Base model | Pythia-1.4B |
+| Hidden width | 2,048 |
+| Attachment | GPT-NeoX layer 23 |
+| Parameters | 2,707,464 |
+| Canonical protocol | `pcm-canonical-p-v1` |
+| Artifact SHA-256 | `72ef68d07ee27c37b90432d34d4be5c2c280ae1bcb08236e37a0e458c054d8d7` |
+
+The TTL maps model hidden states to factorized canonical queries and selected canonical values back to model-hidden residuals. Its gate is conditioned on current hidden state, translated P, and canonical route features. The frozen Pythia base receives zero gradients.
+
+The controlled 128-slot benchmark recorded 100% held-out state generation. In the matched causal test, changing only canonical P changed the answer. Wrong, historical, invalidated, router-disabled, and TTL-disabled conditions restored frozen candidate logits. This proves the tested Pythia path can consume internal state, not that every model can.
+
+The artifact contains adapter tensors and compatibility metadata. It contains no base-model weights, P-cache state, conversation state, prompt tokens, KV, or optimizer state.
+
+This evidence is specific to Pythia-1.4B. The tiny GPT-2 test is structural and does not establish trained semantic portability to another model family.
diff --git a/artifacts/active-system-audit.json b/artifacts/active-system-audit.json
new file mode 100644
index 0000000000000000000000000000000000000000..e63e7d311ca48669e6ced7e232fd245cc89b220a
--- /dev/null
+++ b/artifacts/active-system-audit.json
@@ -0,0 +1,979 @@
+{
+ "cache_router_profile": {
+ "1024": {
+ "context_query_cpu_seconds": 0.00026123900000030176,
+ "context_query_wall_seconds": 0.00026156400053878315,
+ "correct": true,
+ "fixed_allocation_bytes": 2151424,
+ "router_index_bytes": 541696,
+ "routing_cpu_seconds": 0.0001990224999999235,
+ "routing_wall_seconds": 0.0001980755005206447,
+ "selected_index": 1023,
+ "slot_hydration_cpu_seconds": 0.2316533870000006,
+ "slot_hydration_wall_seconds": 0.23300263999772142
+ },
+ "128": {
+ "context_query_cpu_seconds": 0.00026496599999958903,
+ "context_query_wall_seconds": 0.0002643149982759496,
+ "correct": true,
+ "fixed_allocation_bytes": 268928,
+ "router_index_bytes": 67712,
+ "routing_cpu_seconds": 0.00017257199999987094,
+ "routing_wall_seconds": 0.00017198149907926563,
+ "selected_index": 127,
+ "slot_hydration_cpu_seconds": 0.029635333999999958,
+ "slot_hydration_wall_seconds": 0.0301195789979829
+ },
+ "256": {
+ "context_query_cpu_seconds": 0.00026626349999947507,
+ "context_query_wall_seconds": 0.0002665029987838352,
+ "correct": true,
+ "fixed_allocation_bytes": 537856,
+ "router_index_bytes": 135424,
+ "routing_cpu_seconds": 0.0001784940000004731,
+ "routing_wall_seconds": 0.00017715650028549135,
+ "selected_index": 255,
+ "slot_hydration_cpu_seconds": 0.05884093700000026,
+ "slot_hydration_wall_seconds": 0.059160028999031056
+ },
+ "512": {
+ "context_query_cpu_seconds": 0.00026109349999980935,
+ "context_query_wall_seconds": 0.00026036700000986457,
+ "correct": true,
+ "fixed_allocation_bytes": 1075712,
+ "router_index_bytes": 270848,
+ "routing_cpu_seconds": 0.00018318249999982328,
+ "routing_wall_seconds": 0.0001824114988266956,
+ "selected_index": 511,
+ "slot_hydration_cpu_seconds": 0.11655074100000018,
+ "slot_hydration_wall_seconds": 0.1174167370008945
+ },
+ "64": {
+ "context_query_cpu_seconds": 0.000264374499999942,
+ "context_query_wall_seconds": 0.00026382900068711024,
+ "correct": true,
+ "fixed_allocation_bytes": 134464,
+ "router_index_bytes": 33856,
+ "routing_cpu_seconds": 0.00017868450000024794,
+ "routing_wall_seconds": 0.00017796800057112705,
+ "selected_index": 63,
+ "slot_hydration_cpu_seconds": 0.014785765000000062,
+ "slot_hydration_wall_seconds": 0.014910114001395414
+ }
+ },
+ "dependency_map": {
+ "active_module_imports": {
+ "__init__": [
+ "pcm.planner.cache",
+ "pcm.planner.canonical",
+ "pcm.planner.personality",
+ "pcm.planner.pythia_split_translate",
+ "pcm.planner.split_translator"
+ ],
+ "cache": [],
+ "canonical": [
+ "pcm.planner.cache"
+ ],
+ "personality": [
+ "pcm.planner.cache",
+ "pcm.planner.canonical",
+ "pcm.planner.representation",
+ "pcm.planner.split_translator"
+ ],
+ "personality_eval": [
+ "pcm.planner.canonical",
+ "pcm.planner.personality",
+ "pcm.planner.pythia_split_translate",
+ "pcm.planner.representation",
+ "pcm.planner.split_translator"
+ ],
+ "personality_profile": [
+ "pcm.planner.canonical",
+ "pcm.planner.personality",
+ "pcm.planner.representation"
+ ],
+ "pythia_split_translate": [
+ "pcm.planner.canonical",
+ "pcm.planner.split_translator"
+ ],
+ "representation": [],
+ "split_translator": [
+ "pcm.planner.cache",
+ "pcm.planner.canonical"
+ ],
+ "split_translator_eval": [
+ "pcm.planner.cache",
+ "pcm.planner.canonical",
+ "pcm.planner.pythia_split_translate",
+ "pcm.planner.representation",
+ "pcm.planner.split_translator"
+ ]
+ },
+ "archive_dependencies": [],
+ "canonical_stores_model_hidden_vectors": false,
+ "canonical_stores_model_token_ids": false
+ },
+ "experiment": "active-planner-cache-full-audit-v1",
+ "matched_e2e_evidence": {
+ "failure_attribution_fields": "storage validity, selected route/rank, acceptance, gate, candidate logits, generated token, KL, latency, VRAM",
+ "matched_cuda_attribution": {
+ "active_memory": {
+ "context_creative": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 990,
+ "package_disk_bytes": 57344
+ },
+ "context_technical": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 996,
+ "package_disk_bytes": 57344
+ },
+ "irrelevant": {
+ "canonical_store_bytes": 0,
+ "loaded_entries": 0,
+ "logical_disk_bytes_read": 0,
+ "package_disk_bytes": 57344
+ },
+ "low_confidence": {
+ "canonical_store_bytes": 0,
+ "loaded_entries": 0,
+ "logical_disk_bytes_read": 0,
+ "package_disk_bytes": 57344
+ },
+ "package_a": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 787,
+ "package_disk_bytes": 57344
+ },
+ "package_b": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 783,
+ "package_disk_bytes": 57344
+ }
+ },
+ "adapter_training_performed": false,
+ "base_model": "pythia-1.4b",
+ "base_parameters_with_grad": 0,
+ "base_training_performed": false,
+ "candidate_accuracy": {
+ "context_creative_bob": 1.0,
+ "context_technical_alice": 1.0,
+ "irrelevant_matches_base": 1.0,
+ "package_a_alice": 1.0,
+ "package_b_bob": 1.0
+ },
+ "canonical_probe": {
+ "canonical_decode_accuracy": {
+ "entity": 1.0,
+ "metadata": 1.0,
+ "relation": 1.0,
+ "value": 1.0
+ },
+ "hard_negative_accuracy": {
+ "historical": 1.0,
+ "wrong_entity": 0.97265625,
+ "wrong_value": 1.0
+ },
+ "held_out_combinations": 256,
+ "p_only_state_recovery": 0.97265625,
+ "permutation_stability": 1.0,
+ "permutations_per_combination": 8,
+ "slot_width": 512,
+ "total_held_out_combinations": 519,
+ "train_combinations": 2073,
+ "training_loss_first": 7.440117835998535,
+ "training_loss_last": 0.04114125296473503
+ },
+ "canonical_representation_reconstructed_from_fixed_existing_recipe": true,
+ "causal": {
+ "frozen_base": {
+ "active_state_vram_bytes": 0,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": null,
+ "selected_index": null,
+ "selected_state": null
+ },
+ "historical": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": 5.400981426239014,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "invalidated": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": null,
+ "selected_index": null,
+ "selected_state": null
+ },
+ "p_cache_only": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": -0.30040669441223145,
+ "selected_index": 0,
+ "selected_state": "current-task"
+ },
+ "p_cache_plus_p_package": {
+ "active_state_vram_bytes": 2154,
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 1,
+ "selected_state": "user"
+ },
+ "p_package_a_relevant": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "p_package_b_relevant": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 14.0,
+ "alice_probability": 8.776464277548968e-11,
+ "bob_logit": 37.15625,
+ "bob_probability": 0.9998875856399536,
+ "gate": 0.9605370163917542,
+ "generated": " Bob",
+ "kl_from_base": 9.265372276306152,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "p_package_context_creative": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 14.0,
+ "alice_probability": 8.776464277548968e-11,
+ "bob_logit": 37.15625,
+ "bob_probability": 0.9998875856399536,
+ "gate": 0.9605370163917542,
+ "generated": " Bob",
+ "kl_from_base": 9.265372276306152,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "p_package_context_technical": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "p_package_contradictory_low_confidence": {
+ "active_state_vram_bytes": 0,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": null,
+ "selected_index": null,
+ "selected_state": null
+ },
+ "p_package_irrelevant": {
+ "active_state_vram_bytes": 0,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": null,
+ "selected_index": null,
+ "selected_state": null
+ },
+ "router_disabled": {
+ "active_state_vram_bytes": 0,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": null,
+ "selected_index": null,
+ "selected_state": null
+ },
+ "translator_disabled": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "translator_oracle_route": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "wrong_entity": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": -0.30040669441223145,
+ "selected_index": 0,
+ "selected_state": "someone-else"
+ },
+ "wrong_relation": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": 4.078105926513672,
+ "selected_index": 0,
+ "selected_state": "user"
+ }
+ },
+ "experiment": "active-system-cuda-failure-attribution-v1",
+ "extra_prompt_tokens": 0,
+ "full_package_uploaded_to_cuda": false,
+ "inactive_package_entries": 5,
+ "inactive_vram_delta_bytes": 0,
+ "latency_seconds": {
+ "frozen_base": 0.0174486715994135,
+ "historical": 0.019347908600320807,
+ "invalidated": 0.01760459740035003,
+ "p_cache_only": 0.019043124400195666,
+ "p_cache_plus_p_package": 0.019569637600216083,
+ "p_package_a_relevant": 0.019282742799987318,
+ "p_package_b_relevant": 0.018934362999425504,
+ "p_package_context_creative": 0.01985094979972928,
+ "p_package_context_technical": 0.019666298399533842,
+ "p_package_contradictory_low_confidence": 0.0175312320003286,
+ "p_package_irrelevant": 0.01746072020032443,
+ "router_disabled": 0.017419308600074145,
+ "translator_disabled": 0.01867739240042283,
+ "translator_oracle_route": 0.018915273399761644,
+ "wrong_entity": 0.019296769200445853,
+ "wrong_relation": 0.018960535999940475
+ },
+ "natural_interaction": {
+ "frozen_base": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_cache_only": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_cache_plus_p_package": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_contradictory_low_confidence": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_irrelevant": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_relevant": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ }
+ },
+ "relevant_personality_chat": {
+ "base_target_loss": 8.359650611877441,
+ "generated": " Alice",
+ "package_target_loss": -0.0,
+ "target": "Alice",
+ "target_accuracy": 1.0
+ },
+ "source_tokens_in_recent_kv": 0
+ },
+ "natural_rp": {
+ "base_loss": 5.583366870880127,
+ "conditions": {
+ "base": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "invalidated": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "irrelevant": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "wrong_entity": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ }
+ },
+ "relevant_state_generation_sample": " Alice"
+ },
+ "personality_counterfactual": {
+ "frozen_base": {
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08
+ },
+ "p_cache_only": {
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08
+ },
+ "p_cache_plus_p_package": {
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441
+ },
+ "p_package_a_relevant": {
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441
+ },
+ "p_package_b_relevant": {
+ "alice_logit": 14.0,
+ "alice_probability": 8.776464277548968e-11,
+ "bob_logit": 37.15625,
+ "bob_probability": 0.9998875856399536,
+ "gate": 0.9605370163917542,
+ "generated": " Bob",
+ "kl_from_base": 9.265372276306152
+ },
+ "p_package_context_creative": {
+ "alice_logit": 14.0,
+ "alice_probability": 8.776464277548968e-11,
+ "bob_logit": 37.15625,
+ "bob_probability": 0.9998875856399536,
+ "gate": 0.9605370163917542,
+ "generated": " Bob",
+ "kl_from_base": 9.265372276306152
+ },
+ "p_package_context_technical": {
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441
+ },
+ "p_package_contradictory_low_confidence": {
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08
+ },
+ "p_package_irrelevant": {
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08
+ }
+ },
+ "personality_natural_rp": {
+ "frozen_base": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_cache_only": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_cache_plus_p_package": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_contradictory_low_confidence": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_irrelevant": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_relevant": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ }
+ },
+ "source_artifacts": [
+ "artifacts/phase-b-split-translator.json",
+ "artifacts/phase-b-personality-package.json"
+ ],
+ "state_ablations": {
+ "full_system_with_preservation": {
+ "full_token_accuracy": 1.0,
+ "gate_activation": 0.9820089340209961,
+ "state_candidate_accuracy": 1.0
+ },
+ "router_only": {
+ "active_vram_overhead_bytes": 11035444,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.10144930460010074,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "router_plus_translator_plus_gate": {
+ "historical_state_kl": 9.595059236744419e-05,
+ "rp_kl": -3.993045538663864e-08,
+ "rp_loss": 5.583366870880127,
+ "state_candidate_accuracy": 1.0,
+ "state_loss": 1.8655489839147776e-05,
+ "wrong_state_kl": 9.595059236744419e-05
+ },
+ "router_plus_translator_without_gate": {
+ "state_candidate_accuracy": 1.0
+ },
+ "translator_only_oracle_routing": {
+ "state_candidate_accuracy": 1.0
+ }
+ },
+ "state_counterfactual": {
+ "disabled": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p1_silver_alice": {
+ "alice_logit": 40.09375,
+ "alice_probability": 1.0,
+ "bob_logit": 14.3125,
+ "bob_probability": 6.3583643558629e-12,
+ "gate": 0.9964228272438049,
+ "generated": " Alice"
+ },
+ "p2_silver_bob": {
+ "alice_logit": 13.0234375,
+ "alice_probability": 4.473376963992637e-12,
+ "bob_logit": 39.15625,
+ "bob_probability": 0.9999349117279053,
+ "gate": 0.9842994213104248,
+ "generated": " Bob"
+ },
+ "p3_gold_alice": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p4_silver_historical": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p4_silver_invalidated": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ }
+ }
+ },
+ "ppkg_scaling": {
+ "100": {
+ "build_seconds": 0.005182955999771366,
+ "candidate_headers": 4,
+ "canonical_conversion_cpu_seconds": 0.0005506600000000361,
+ "canonical_conversion_wall_seconds": 0.0005500090010173153,
+ "checksum_seconds": 0.0010304470015398692,
+ "db_open_seconds": 0.00020484400010900572,
+ "disk_bytes": 118784,
+ "entries_loaded": 4,
+ "inactive_vram_bytes": 0,
+ "logical_bytes_read": 2789,
+ "python_peak_allocation_bytes": 9086,
+ "query_plan": {
+ "scope": [
+ "(4, 0, 54, 'SEARCH entries USING COVERING INDEX entry_scope_route (status=? AND scope=?)')"
+ ],
+ "subject": [
+ "(4, 0, 54, 'SEARCH entries USING COVERING INDEX entry_subject_route (status=? AND subject=?)')"
+ ]
+ },
+ "routing_header_cpu_seconds": 0.002198153000000147,
+ "routing_header_wall_seconds": 0.002207458997872891,
+ "row_hydration_cpu_seconds": 0.00011678899999978398,
+ "row_hydration_wall_seconds": 0.00011649800217128359
+ },
+ "1000": {
+ "build_seconds": 0.030133395001030294,
+ "candidate_headers": 33,
+ "canonical_conversion_cpu_seconds": 0.0004903580000004126,
+ "canonical_conversion_wall_seconds": 0.0004913600023428444,
+ "checksum_seconds": 0.008523084998159902,
+ "db_open_seconds": 0.0002957940014312044,
+ "disk_bytes": 638976,
+ "entries_loaded": 4,
+ "inactive_vram_bytes": 0,
+ "logical_bytes_read": 6855,
+ "python_peak_allocation_bytes": 19832,
+ "query_plan": {
+ "scope": [
+ "(4, 0, 54, 'SEARCH entries USING COVERING INDEX entry_scope_route (status=? AND scope=?)')"
+ ],
+ "subject": [
+ "(4, 0, 54, 'SEARCH entries USING COVERING INDEX entry_subject_route (status=? AND subject=?)')"
+ ]
+ },
+ "routing_header_cpu_seconds": 0.014248357999999683,
+ "routing_header_wall_seconds": 0.014324849998956779,
+ "row_hydration_cpu_seconds": 0.00011794099999917762,
+ "row_hydration_wall_seconds": 0.0001179409991891589
+ },
+ "10000": {
+ "build_seconds": 0.29831144099807716,
+ "candidate_headers": 130,
+ "canonical_conversion_cpu_seconds": 0.0005051340000008508,
+ "canonical_conversion_wall_seconds": 0.0005099939990031999,
+ "checksum_seconds": 0.08329692499683006,
+ "db_open_seconds": 0.00028353100060485303,
+ "disk_bytes": 5894144,
+ "entries_loaded": 4,
+ "inactive_vram_bytes": 0,
+ "logical_bytes_read": 20616,
+ "python_peak_allocation_bytes": 75123,
+ "query_plan": {
+ "scope": [
+ "(4, 0, 54, 'SEARCH entries USING COVERING INDEX entry_scope_route (status=? AND scope=?)')"
+ ],
+ "subject": [
+ "(4, 0, 54, 'SEARCH entries USING COVERING INDEX entry_subject_route (status=? AND subject=?)')"
+ ]
+ },
+ "routing_header_cpu_seconds": 0.05690157699999965,
+ "routing_header_wall_seconds": 0.05726086499998928,
+ "row_hydration_cpu_seconds": 0.00011336200000044983,
+ "row_hydration_wall_seconds": 0.00011310199988656677
+ },
+ "100000": {
+ "build_seconds": 3.1510163159982767,
+ "candidate_headers": 152,
+ "canonical_conversion_cpu_seconds": 0.0004911590000009625,
+ "canonical_conversion_wall_seconds": 0.0004947460001858417,
+ "checksum_seconds": 0.8238876559989876,
+ "db_open_seconds": 0.000549067000974901,
+ "disk_bytes": 58720256,
+ "entries_loaded": 4,
+ "inactive_vram_bytes": 0,
+ "logical_bytes_read": 23715,
+ "python_peak_allocation_bytes": 87891,
+ "query_plan": {
+ "scope": [
+ "(4, 0, 54, 'SEARCH entries USING COVERING INDEX entry_scope_route (status=? AND scope=?)')"
+ ],
+ "subject": [
+ "(4, 0, 54, 'SEARCH entries USING COVERING INDEX entry_subject_route (status=? AND subject=?)')"
+ ]
+ },
+ "routing_header_cpu_seconds": 0.06670220700000051,
+ "routing_header_wall_seconds": 0.06710040300094988,
+ "row_hydration_cpu_seconds": 0.00011785099999883641,
+ "row_hydration_wall_seconds": 0.00011752000136766583
+ }
+ },
+ "process_peak_rss_bytes": 1001316352,
+ "router_scaling": {
+ "attribution": "The pre-audit CanonicalPStore allowed semantic merge across different entity/relation identities. Repeated factorized vectors therefore aliased slots at 256/512 even for oracle queries. Identity-constrained merge removes that storage corruption; the matched post-fix canonical router remains 100% through 1024. Linear scan affects latency but did not cause the accuracy loss.",
+ "immutable_pre_fix_baseline": {
+ "128": 1.0,
+ "256": 0.949999988079071,
+ "512": 0.8500000238418579
+ },
+ "measurements": {
+ "1024": {
+ "entity_confusions": 0,
+ "exact_query_anchor_collisions": 0,
+ "failure_examples": [],
+ "historical_confusions": 0,
+ "latency_seconds": 0.0006750829998054542,
+ "mrr": 1.0,
+ "relation_confusions": 0,
+ "top1_accuracy": 1.0,
+ "top4_recall": 1.0
+ },
+ "128": {
+ "entity_confusions": 0,
+ "exact_query_anchor_collisions": 0,
+ "failure_examples": [],
+ "historical_confusions": 0,
+ "latency_seconds": 0.0003642720002972055,
+ "mrr": 1.0,
+ "relation_confusions": 0,
+ "top1_accuracy": 1.0,
+ "top4_recall": 1.0
+ },
+ "20": {
+ "entity_confusions": 0,
+ "exact_query_anchor_collisions": 0,
+ "failure_examples": [],
+ "historical_confusions": 0,
+ "latency_seconds": 0.0003294770031061489,
+ "mrr": 1.0,
+ "relation_confusions": 0,
+ "top1_accuracy": 1.0,
+ "top4_recall": 1.0
+ },
+ "256": {
+ "entity_confusions": 0,
+ "exact_query_anchor_collisions": 0,
+ "failure_examples": [],
+ "historical_confusions": 0,
+ "latency_seconds": 0.00039848600135883316,
+ "mrr": 1.0,
+ "relation_confusions": 0,
+ "top1_accuracy": 1.0,
+ "top4_recall": 1.0
+ },
+ "4": {
+ "entity_confusions": 0,
+ "exact_query_anchor_collisions": 0,
+ "failure_examples": [],
+ "historical_confusions": 0,
+ "latency_seconds": 0.00035979299718746915,
+ "mrr": 1.0,
+ "relation_confusions": 0,
+ "top1_accuracy": 1.0,
+ "top4_recall": 1.0
+ },
+ "512": {
+ "entity_confusions": 0,
+ "exact_query_anchor_collisions": 0,
+ "failure_examples": [],
+ "historical_confusions": 0,
+ "latency_seconds": 0.0004960180012858473,
+ "mrr": 1.0,
+ "relation_confusions": 0,
+ "top1_accuracy": 1.0,
+ "top4_recall": 1.0
+ },
+ "64": {
+ "entity_confusions": 0,
+ "exact_query_anchor_collisions": 0,
+ "failure_examples": [],
+ "historical_confusions": 0,
+ "latency_seconds": 0.00033151000025100075,
+ "mrr": 1.0,
+ "relation_confusions": 0,
+ "top1_accuracy": 1.0,
+ "top4_recall": 1.0
+ }
+ }
+ },
+ "timestamp": "2026-08-22T00:00:00+00:00",
+ "training_performed": false,
+ "translate_profile": {
+ "cpu": {
+ "parameter_bytes": 10829856,
+ "parameter_count": 2707464,
+ "translation_and_gate_cpu_seconds": 0.003563432500000019,
+ "translation_and_gate_wall_seconds": 0.0006066100013413234
+ },
+ "cuda": {
+ "active_vram_delta_bytes": 8914432,
+ "bytes_copied_to_cuda": 164096,
+ "copy_wall_seconds": 0.0005022000004828442,
+ "peak_vram_delta_bytes": 9441280,
+ "translation_and_gate_wall_seconds": 0.0012042837000384072
+ }
+ }
+}
diff --git a/artifacts/active-system-cuda-attribution.json b/artifacts/active-system-cuda-attribution.json
new file mode 100644
index 0000000000000000000000000000000000000000..9b15026474df8b771fae653a003c8188948c721b
--- /dev/null
+++ b/artifacts/active-system-cuda-attribution.json
@@ -0,0 +1,387 @@
+{
+ "active_memory": {
+ "context_creative": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 990,
+ "package_disk_bytes": 57344
+ },
+ "context_technical": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 996,
+ "package_disk_bytes": 57344
+ },
+ "irrelevant": {
+ "canonical_store_bytes": 0,
+ "loaded_entries": 0,
+ "logical_disk_bytes_read": 0,
+ "package_disk_bytes": 57344
+ },
+ "low_confidence": {
+ "canonical_store_bytes": 0,
+ "loaded_entries": 0,
+ "logical_disk_bytes_read": 0,
+ "package_disk_bytes": 57344
+ },
+ "package_a": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 787,
+ "package_disk_bytes": 57344
+ },
+ "package_b": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 783,
+ "package_disk_bytes": 57344
+ }
+ },
+ "adapter_training_performed": false,
+ "base_model": "pythia-1.4b",
+ "base_parameters_with_grad": 0,
+ "base_training_performed": false,
+ "candidate_accuracy": {
+ "context_creative_bob": 1.0,
+ "context_technical_alice": 1.0,
+ "irrelevant_matches_base": 1.0,
+ "package_a_alice": 1.0,
+ "package_b_bob": 1.0
+ },
+ "canonical_probe": {
+ "canonical_decode_accuracy": {
+ "entity": 1.0,
+ "metadata": 1.0,
+ "relation": 1.0,
+ "value": 1.0
+ },
+ "hard_negative_accuracy": {
+ "historical": 1.0,
+ "wrong_entity": 0.97265625,
+ "wrong_value": 1.0
+ },
+ "held_out_combinations": 256,
+ "p_only_state_recovery": 0.97265625,
+ "permutation_stability": 1.0,
+ "permutations_per_combination": 8,
+ "slot_width": 512,
+ "total_held_out_combinations": 519,
+ "train_combinations": 2073,
+ "training_loss_first": 7.440117835998535,
+ "training_loss_last": 0.04114125296473503
+ },
+ "canonical_representation_reconstructed_from_fixed_existing_recipe": true,
+ "causal": {
+ "frozen_base": {
+ "active_state_vram_bytes": 0,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": null,
+ "selected_index": null,
+ "selected_state": null
+ },
+ "historical": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": 5.400981426239014,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "invalidated": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": null,
+ "selected_index": null,
+ "selected_state": null
+ },
+ "p_cache_only": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": -0.30040669441223145,
+ "selected_index": 0,
+ "selected_state": "current-task"
+ },
+ "p_cache_plus_p_package": {
+ "active_state_vram_bytes": 2154,
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 1,
+ "selected_state": "user"
+ },
+ "p_package_a_relevant": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "p_package_b_relevant": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 14.0,
+ "alice_probability": 8.776464277548968e-11,
+ "bob_logit": 37.15625,
+ "bob_probability": 0.9998875856399536,
+ "gate": 0.9605370163917542,
+ "generated": " Bob",
+ "kl_from_base": 9.265372276306152,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "p_package_context_creative": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 14.0,
+ "alice_probability": 8.776464277548968e-11,
+ "bob_logit": 37.15625,
+ "bob_probability": 0.9998875856399536,
+ "gate": 0.9605370163917542,
+ "generated": " Bob",
+ "kl_from_base": 9.265372276306152,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "p_package_context_technical": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "p_package_contradictory_low_confidence": {
+ "active_state_vram_bytes": 0,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": null,
+ "selected_index": null,
+ "selected_state": null
+ },
+ "p_package_irrelevant": {
+ "active_state_vram_bytes": 0,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": null,
+ "selected_index": null,
+ "selected_state": null
+ },
+ "router_disabled": {
+ "active_state_vram_bytes": 0,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": null,
+ "selected_index": null,
+ "selected_state": null
+ },
+ "translator_disabled": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "translator_oracle_route": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441,
+ "router_accepted": true,
+ "router_score": 5.590433597564697,
+ "selected_index": 0,
+ "selected_state": "user"
+ },
+ "wrong_entity": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": -0.30040669441223145,
+ "selected_index": 0,
+ "selected_state": "someone-else"
+ },
+ "wrong_relation": {
+ "active_state_vram_bytes": 1077,
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08,
+ "router_accepted": false,
+ "router_score": 4.078105926513672,
+ "selected_index": 0,
+ "selected_state": "user"
+ }
+ },
+ "experiment": "active-system-cuda-failure-attribution-v1",
+ "extra_prompt_tokens": 0,
+ "full_package_uploaded_to_cuda": false,
+ "inactive_package_entries": 5,
+ "inactive_vram_delta_bytes": 0,
+ "latency_seconds": {
+ "frozen_base": 0.0174486715994135,
+ "historical": 0.019347908600320807,
+ "invalidated": 0.01760459740035003,
+ "p_cache_only": 0.019043124400195666,
+ "p_cache_plus_p_package": 0.019569637600216083,
+ "p_package_a_relevant": 0.019282742799987318,
+ "p_package_b_relevant": 0.018934362999425504,
+ "p_package_context_creative": 0.01985094979972928,
+ "p_package_context_technical": 0.019666298399533842,
+ "p_package_contradictory_low_confidence": 0.0175312320003286,
+ "p_package_irrelevant": 0.01746072020032443,
+ "router_disabled": 0.017419308600074145,
+ "translator_disabled": 0.01867739240042283,
+ "translator_oracle_route": 0.018915273399761644,
+ "wrong_entity": 0.019296769200445853,
+ "wrong_relation": 0.018960535999940475
+ },
+ "natural_interaction": {
+ "frozen_base": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_cache_only": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_cache_plus_p_package": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_contradictory_low_confidence": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_irrelevant": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_relevant": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ }
+ },
+ "relevant_personality_chat": {
+ "base_target_loss": 8.359650611877441,
+ "generated": " Alice",
+ "package_target_loss": -0.0,
+ "target": "Alice",
+ "target_accuracy": 1.0
+ },
+ "source_tokens_in_recent_kv": 0
+}
diff --git a/artifacts/canonical-p-v1.router b/artifacts/canonical-p-v1.router
new file mode 100644
index 0000000000000000000000000000000000000000..bc7c437fad20e910c1c3cbb2316d54efacb46a0f
Binary files /dev/null and b/artifacts/canonical-p-v1.router differ
diff --git a/artifacts/debug-actions-profile.json b/artifacts/debug-actions-profile.json
new file mode 100644
index 0000000000000000000000000000000000000000..4d2ac281bc7b89b37b857f3809652f9b598ef4fc
--- /dev/null
+++ b/artifacts/debug-actions-profile.json
@@ -0,0 +1,245 @@
+{
+ "experiment": "planner-cache-debug-actions-profile-v1",
+ "page_limit": 100,
+ "personality": {
+ "100": {
+ "active_entries": 100,
+ "after_bounded": {
+ "action": {
+ "cpu_seconds": 0.009221641999999974,
+ "peak_python_allocation_bytes": 196984,
+ "wall_seconds": 0.009249514012481086
+ },
+ "hydrated_entries": 100,
+ "json_serialization": {
+ "cpu_seconds": 0.003703828000000353,
+ "peak_python_allocation_bytes": 121302,
+ "wall_seconds": 0.003694181010359898
+ },
+ "response_bytes": 60318,
+ "total_active": 100,
+ "truncated": false
+ },
+ "before_unbounded": {
+ "action": {
+ "cpu_seconds": 0.009142955000000175,
+ "peak_python_allocation_bytes": 196591,
+ "wall_seconds": 0.009165266004856676
+ },
+ "hydrated_entries": 100,
+ "json_serialization": {
+ "cpu_seconds": 0.003326453000000118,
+ "peak_python_allocation_bytes": 121106,
+ "wall_seconds": 0.003322666001622565
+ },
+ "response_bytes": 60220
+ }
+ },
+ "1000": {
+ "active_entries": 1000,
+ "after_bounded": {
+ "action": {
+ "cpu_seconds": 0.009862894999999927,
+ "peak_python_allocation_bytes": 196288,
+ "wall_seconds": 0.009891455003526062
+ },
+ "hydrated_entries": 100,
+ "json_serialization": {
+ "cpu_seconds": 0.003532476999999812,
+ "peak_python_allocation_bytes": 121302,
+ "wall_seconds": 0.0035344719944987446
+ },
+ "response_bytes": 60318,
+ "total_active": 1000,
+ "truncated": true
+ },
+ "before_unbounded": {
+ "action": {
+ "cpu_seconds": 0.09285407999999995,
+ "peak_python_allocation_bytes": 1928216,
+ "wall_seconds": 0.09317578599439003
+ },
+ "hydrated_entries": 1000,
+ "json_serialization": {
+ "cpu_seconds": 0.03305002400000001,
+ "peak_python_allocation_bytes": 1211130,
+ "wall_seconds": 0.033163794010761194
+ },
+ "response_bytes": 605232
+ }
+ },
+ "10000": {
+ "active_entries": 10000,
+ "after_bounded": {
+ "action": {
+ "cpu_seconds": 0.010486288000000066,
+ "peak_python_allocation_bytes": 196288,
+ "wall_seconds": 0.010522566008148715
+ },
+ "hydrated_entries": 100,
+ "json_serialization": {
+ "cpu_seconds": 0.003585478999999836,
+ "peak_python_allocation_bytes": 121304,
+ "wall_seconds": 0.00358573799894657
+ },
+ "response_bytes": 60319,
+ "total_active": 10000,
+ "truncated": true
+ },
+ "before_unbounded": {
+ "action": {
+ "cpu_seconds": 0.9330532409999996,
+ "peak_python_allocation_bytes": 17663949,
+ "wall_seconds": 0.9362227259989595
+ },
+ "hydrated_entries": 10000,
+ "json_serialization": {
+ "cpu_seconds": 0.34413546300000064,
+ "peak_python_allocation_bytes": 12161028,
+ "wall_seconds": 0.3450379890127806
+ },
+ "response_bytes": 6080181
+ }
+ },
+ "100000": {
+ "active_entries": 100000,
+ "after_bounded": {
+ "action": {
+ "cpu_seconds": 0.021668488999999624,
+ "peak_python_allocation_bytes": 196288,
+ "wall_seconds": 0.02177674999984447
+ },
+ "hydrated_entries": 100,
+ "json_serialization": {
+ "cpu_seconds": 0.0037721280000013735,
+ "peak_python_allocation_bytes": 121306,
+ "wall_seconds": 0.003769953007576987
+ },
+ "response_bytes": 60320,
+ "total_active": 100000,
+ "truncated": true
+ },
+ "before_unbounded": {
+ "action": {
+ "cpu_seconds": 9.748923195,
+ "peak_python_allocation_bytes": 173110748,
+ "wall_seconds": 9.787373771992861
+ },
+ "hydrated_entries": 100000,
+ "json_serialization": {
+ "cpu_seconds": 3.3711310019999985,
+ "peak_python_allocation_bytes": 122015586,
+ "wall_seconds": 3.384944927005563
+ },
+ "response_bytes": 61007460
+ }
+ }
+ },
+ "personality_query_plans": {
+ "active_count": [
+ "SEARCH entries USING COVERING INDEX entry_relationship_route (status=?)"
+ ],
+ "bounded_page": [
+ "SEARCH entries USING INDEX entry_relationship_route (status=?)",
+ "USE TEMP B-TREE FOR ORDER BY"
+ ]
+ },
+ "state": {
+ "0": {
+ "active_entries": 0,
+ "bounded_by_configured_capacity": true,
+ "configured_slots": 1,
+ "json_serialization": {
+ "cpu_seconds": 6.78270000000758e-05,
+ "peak_python_allocation_bytes": 1259,
+ "wall_seconds": 5.7437995565123856e-05
+ },
+ "response_bytes": 2,
+ "snapshot": {
+ "cpu_seconds": 0.00010548699999990419,
+ "peak_python_allocation_bytes": 432,
+ "wall_seconds": 8.957799582276493e-05
+ }
+ },
+ "1024": {
+ "active_entries": 1024,
+ "bounded_by_configured_capacity": true,
+ "configured_slots": 1024,
+ "json_serialization": {
+ "cpu_seconds": 0.030256003999999947,
+ "peak_python_allocation_bytes": 643302,
+ "wall_seconds": 0.030321376005304046
+ },
+ "response_bytes": 321318,
+ "snapshot": {
+ "cpu_seconds": 0.04200101300000014,
+ "peak_python_allocation_bytes": 716728,
+ "wall_seconds": 0.042143614002270624
+ }
+ },
+ "128": {
+ "active_entries": 128,
+ "bounded_by_configured_capacity": true,
+ "configured_slots": 128,
+ "json_serialization": {
+ "cpu_seconds": 0.0038881549999998377,
+ "peak_python_allocation_bytes": 80572,
+ "wall_seconds": 0.0038865509995957837
+ },
+ "response_bytes": 39953,
+ "snapshot": {
+ "cpu_seconds": 0.005286958999999758,
+ "peak_python_allocation_bytes": 86776,
+ "wall_seconds": 0.005294073998811655
+ }
+ },
+ "256": {
+ "active_entries": 256,
+ "bounded_by_configured_capacity": true,
+ "configured_slots": 256,
+ "json_serialization": {
+ "cpu_seconds": 0.0072936459999999315,
+ "peak_python_allocation_bytes": 160958,
+ "wall_seconds": 0.00729949600645341
+ },
+ "response_bytes": 80146,
+ "snapshot": {
+ "cpu_seconds": 0.010555682000000122,
+ "peak_python_allocation_bytes": 173272,
+ "wall_seconds": 0.010585684009129182
+ }
+ },
+ "512": {
+ "active_entries": 512,
+ "bounded_by_configured_capacity": true,
+ "configured_slots": 512,
+ "json_serialization": {
+ "cpu_seconds": 0.013979924000000032,
+ "peak_python_allocation_bytes": 321704,
+ "wall_seconds": 0.014006733996211551
+ },
+ "response_bytes": 160519,
+ "snapshot": {
+ "cpu_seconds": 0.021038871000000015,
+ "peak_python_allocation_bytes": 354200,
+ "wall_seconds": 0.02109747999929823
+ }
+ },
+ "64": {
+ "active_entries": 64,
+ "bounded_by_configured_capacity": true,
+ "configured_slots": 64,
+ "json_serialization": {
+ "cpu_seconds": 0.0018557219999997265,
+ "peak_python_allocation_bytes": 40532,
+ "wall_seconds": 0.0018496399861760437
+ },
+ "response_bytes": 19933,
+ "snapshot": {
+ "cpu_seconds": 0.0027572279999996674,
+ "peak_python_allocation_bytes": 43576,
+ "wall_seconds": 0.0027569780068006366
+ }
+ }
+ }
+}
diff --git a/artifacts/gemma-native-prompt-equivalence.json b/artifacts/gemma-native-prompt-equivalence.json
new file mode 100644
index 0000000000000000000000000000000000000000..1d7abbbcc60b8b8f9a46048c6613ac18eaf5858f
--- /dev/null
+++ b/artifacts/gemma-native-prompt-equivalence.json
@@ -0,0 +1,29 @@
+{
+ "assertions": {
+ "inactive_ltl_has_no_output_control": true,
+ "message_structure_exact": true,
+ "rendered_prompt_exact": true,
+ "token_ids_exact": true
+ },
+ "conditions": {
+ "ltl_logit_bias_present": false,
+ "message_structure_equal": true,
+ "p_active": false
+ },
+ "format": "planner-cache-native-prompt-equivalence-v1",
+ "gateway_inactive": {
+ "message_sha256": "46f75bef337eb9fcb9cbebb77c8836fa48bc0c1c8de68d0ec1aa24631b912e4d",
+ "prompt_sha256": "0f9bac487f601d8f2c33c9097be424b7adc2813930f98993bd72c9f6ae03777b",
+ "token_count": 33,
+ "token_ids_sha256": "0ace5a54d9cc7305143599c4414212a1946494231473b590fad117e6105d5dcb"
+ },
+ "llama_cpp_version": "version: 10276 (6ea215d17)\nbuilt with GNU 16.1.1 for Linux x86_64",
+ "model": "${GEMMA_MODEL}",
+ "model_sha256": "a4c4177f9fd7e3f56522675afb742f079a53f9226195b7db5e9888c872f053da",
+ "raw": {
+ "message_sha256": "46f75bef337eb9fcb9cbebb77c8836fa48bc0c1c8de68d0ec1aa24631b912e4d",
+ "prompt_sha256": "0f9bac487f601d8f2c33c9097be424b7adc2813930f98993bd72c9f6ae03777b",
+ "token_count": 33,
+ "token_ids_sha256": "0ace5a54d9cc7305143599c4414212a1946494231473b590fad117e6105d5dcb"
+ }
+}
diff --git a/artifacts/gemma4-e4b-q8-causal.json b/artifacts/gemma4-e4b-q8-causal.json
new file mode 100644
index 0000000000000000000000000000000000000000..7da04981d7b8f084d0a51909ee9c71556c9bb6dd
--- /dev/null
+++ b/artifacts/gemma4-e4b-q8-causal.json
@@ -0,0 +1,463 @@
+{
+ "adapter": {
+ "active_control_buffer_bytes": 419840,
+ "adapter_parameter_count": 3073,
+ "base_model_optimization": false,
+ "bytes_copied_per_relevant_activation": 10240,
+ "canonical_probe": {
+ "canonical_decode_accuracy": {
+ "entity": 1.0,
+ "metadata": 1.0,
+ "relation": 1.0,
+ "value": 1.0
+ },
+ "hard_negative_accuracy": {
+ "historical": 1.0,
+ "wrong_entity": 0.97265625,
+ "wrong_value": 1.0
+ },
+ "held_out_combinations": 256,
+ "p_only_state_recovery": 0.97265625,
+ "permutation_stability": 1.0,
+ "permutations_per_combination": 8,
+ "slot_width": 512,
+ "total_held_out_combinations": 519,
+ "train_combinations": 2073,
+ "training_loss_first": 7.440117835998535,
+ "training_loss_last": 0.04114125296473503
+ },
+ "config": {
+ "architecture": "canonical_scalar_to_control_vector_v1",
+ "attachment_layer": 41,
+ "canonical_protocol": "pcm-canonical-p-v1",
+ "canonical_width": 512,
+ "format": "pcm-llama-gguf-translate-v1",
+ "llama_build": "version: 10276 (6ea215d17)\nbuilt with GNU 16.1.1 for Linux x86_64",
+ "model_architecture": "gemma4",
+ "model_hidden_width": 2560,
+ "model_id": "Gemma-4-E4B-Uncensored-HauhauCS-Aggressive",
+ "model_layer_count": 42,
+ "model_sha256": "a4c4177f9fd7e3f56522675afb742f079a53f9226195b7db5e9888c872f053da",
+ "runtime": "llama.cpp-control-vector",
+ "supported_value_ids": [
+ 0,
+ 1
+ ],
+ "target_token_ids": [
+ 32858,
+ 15943
+ ]
+ },
+ "extra_prompt_tokens": 0,
+ "fit_observed_strengths": {
+ "Alice": 1.999999761581421,
+ "Bob": -5.000000953674316
+ },
+ "fit_target_strengths": {
+ "Alice": 2.0,
+ "Bob": -5.0
+ },
+ "gate_basis": "universal canonical route acceptance plus supported canonical value",
+ "inactive_vram_bytes": 0,
+ "model_hidden_query_projection": "not available through llama-server and not claimed",
+ "path": "${REPOSITORY_ROOT}/artifacts/gemma4-e4b-q8-llama.translate",
+ "recent_kv_source_token_count": 0,
+ "sha256": "2a25f1641f796d520250d0367dd3b546579a2050c5ef84e2cfff7e598d299785",
+ "size_bytes": 13356,
+ "training_method": "analytic minimum-norm affine fit to frozen token-row direction"
+ },
+ "assertions": {
+ "alice_generation": true,
+ "alice_logit_lift": true,
+ "all_inactive_paths_exact": true,
+ "bob_generation": true,
+ "bob_logit_lift": true,
+ "historical_rejected": true,
+ "invalidated_rejected": true,
+ "natural_irrelevant_exact": true,
+ "server_alice_generation": true,
+ "server_bob_generation": true,
+ "wrong_entity_rejected": true,
+ "wrong_relation_rejected": true
+ },
+ "conditions": {
+ "correct_alice": {
+ "alice_logit": 26.0793858,
+ "alice_probability": 0.389850411,
+ "bob_logit": 0.909999311,
+ "bob_probability": 4.57059089e-12,
+ "gate": 1.0,
+ "generated": " Alice",
+ "generated_token_id": 32858,
+ "kl_from_base": 0.604321331,
+ "latency_ms": 320.229877,
+ "max_abs_logit_difference_from_base": 11.1606493,
+ "router_accepted": true,
+ "router_score": 5.590437412261963,
+ "scale": 1.99999976,
+ "selected_state": {
+ "label": "silver key",
+ "metadata_id": 0,
+ "relation_id": 0,
+ "value_id": 0
+ },
+ "strength": 1.999999761581421
+ },
+ "correct_bob": {
+ "alice_logit": 7.2261529,
+ "alice_probability": 2.57553064e-09,
+ "bob_logit": 26.6650352,
+ "bob_probability": 0.712961469,
+ "gate": 1.0,
+ "generated": " Bob",
+ "generated_token_id": 15943,
+ "kl_from_base": 9.63846737,
+ "latency_ms": 312.696353,
+ "max_abs_logit_difference_from_base": 23.8088799,
+ "router_accepted": true,
+ "router_score": 5.590437412261963,
+ "scale": -5.00000048,
+ "selected_state": {
+ "label": "silver key",
+ "metadata_id": 0,
+ "relation_id": 0,
+ "value_id": 1
+ },
+ "strength": -5.000000476837158
+ },
+ "historical": {
+ "alice_logit": 23.1685543,
+ "alice_probability": 0.0447660284,
+ "bob_logit": 12.0706482,
+ "bob_probability": 6.77936756e-07,
+ "gate": 0.0,
+ "generated": " the",
+ "generated_token_id": 506,
+ "kl_from_base": 0,
+ "latency_ms": 404.050531,
+ "max_abs_logit_difference_from_base": 0,
+ "router_accepted": false,
+ "router_score": 5.40098237991333,
+ "scale": 0,
+ "selected_state": {
+ "label": "silver key",
+ "metadata_id": 2,
+ "relation_id": 0,
+ "value_id": 0
+ },
+ "strength": 0.0
+ },
+ "invalidated": {
+ "alice_logit": 23.1685543,
+ "alice_probability": 0.0447660284,
+ "bob_logit": 12.0706482,
+ "bob_probability": 6.77936756e-07,
+ "gate": 0.0,
+ "generated": " the",
+ "generated_token_id": 506,
+ "kl_from_base": 0,
+ "latency_ms": 338.130937,
+ "max_abs_logit_difference_from_base": 0,
+ "router_accepted": false,
+ "router_score": null,
+ "scale": 0,
+ "selected_state": null,
+ "strength": 0.0
+ },
+ "p_disabled": {
+ "alice_logit": 23.1685543,
+ "alice_probability": 0.0447660284,
+ "bob_logit": 12.0706482,
+ "bob_probability": 6.77936756e-07,
+ "gate": 0.0,
+ "generated": " the",
+ "generated_token_id": 506,
+ "kl_from_base": 0,
+ "latency_ms": 354.148533,
+ "max_abs_logit_difference_from_base": 0,
+ "router_accepted": false,
+ "router_score": null,
+ "scale": 0,
+ "selected_state": null,
+ "strength": 0.0
+ },
+ "router_disabled": {
+ "alice_logit": 23.1685543,
+ "alice_probability": 0.0447660284,
+ "bob_logit": 12.0706482,
+ "bob_probability": 6.77936756e-07,
+ "gate": 0.0,
+ "generated": " the",
+ "generated_token_id": 506,
+ "kl_from_base": 0,
+ "latency_ms": 385.683319,
+ "max_abs_logit_difference_from_base": 0,
+ "router_accepted": false,
+ "router_score": null,
+ "scale": 0,
+ "selected_state": null,
+ "strength": 0.0
+ },
+ "translator_disabled": {
+ "alice_logit": 23.1685543,
+ "alice_probability": 0.0447660284,
+ "bob_logit": 12.0706482,
+ "bob_probability": 6.77936756e-07,
+ "gate": 0.0,
+ "generated": " the",
+ "generated_token_id": 506,
+ "kl_from_base": 0,
+ "latency_ms": 335.948864,
+ "max_abs_logit_difference_from_base": 0,
+ "router_accepted": true,
+ "router_score": 5.590437412261963,
+ "scale": 0,
+ "selected_state": {
+ "label": "silver key",
+ "metadata_id": 0,
+ "relation_id": 0,
+ "value_id": 0
+ },
+ "strength": 0.0
+ },
+ "wrong_entity": {
+ "alice_logit": 23.1685543,
+ "alice_probability": 0.0447660284,
+ "bob_logit": 12.0706482,
+ "bob_probability": 6.77936756e-07,
+ "gate": 0.0,
+ "generated": " the",
+ "generated_token_id": 506,
+ "kl_from_base": 0,
+ "latency_ms": 321.541691,
+ "max_abs_logit_difference_from_base": 0,
+ "router_accepted": false,
+ "router_score": -0.3507542610168457,
+ "scale": 0,
+ "selected_state": {
+ "label": "gold key",
+ "metadata_id": 0,
+ "relation_id": 0,
+ "value_id": 0
+ },
+ "strength": 0.0
+ },
+ "wrong_relation": {
+ "alice_logit": 23.1685543,
+ "alice_probability": 0.0447660284,
+ "bob_logit": 12.0706482,
+ "bob_probability": 6.77936756e-07,
+ "gate": 0.0,
+ "generated": " the",
+ "generated_token_id": 506,
+ "kl_from_base": 0,
+ "latency_ms": 328.731662,
+ "max_abs_logit_difference_from_base": 0,
+ "router_accepted": false,
+ "router_score": 4.078108310699463,
+ "scale": 0,
+ "selected_state": {
+ "label": "silver key",
+ "metadata_id": 0,
+ "relation_id": 1,
+ "value_id": 0
+ },
+ "strength": 0.0
+ }
+ },
+ "experiment": "gemma4-e4b-q8-llama-translate-causal-v1",
+ "llama_cpp": {
+ "cli": "${LOCAL_PATH}",
+ "control_vector_generator_failure": "stock generator asserted because Gemma4 did not expose n_layers minus one callback tensors",
+ "root": "${LOCAL_PATH}",
+ "runtime_attachment": "public llama_set_adapter_cvec API and llama-server control-vector path",
+ "server": "${LOCAL_PATH}",
+ "version": "version: 10276 (6ea215d17)\nbuilt with GNU 16.1.1 for Linux x86_64"
+ },
+ "model": {
+ "architecture": "gemma4",
+ "base_modified": false,
+ "block_count": 42,
+ "context_length": 131072,
+ "embedding_length": 2560,
+ "file_type": 7,
+ "name": "Gemma-4-E4B-Uncensored-HauhauCS-Aggressive",
+ "path": "${GEMMA_MODEL}",
+ "quantized_gguf": true,
+ "sha256": "a4c4177f9fd7e3f56522675afb742f079a53f9226195b7db5e9888c872f053da",
+ "sha256_seconds": 5.7932635489996755,
+ "size_bytes": 8133226464,
+ "size_label": "7.5B",
+ "tensor_count": 720
+ },
+ "natural_rp": {
+ "conditions": [
+ {
+ "alice_logit": 21.079689,
+ "alice_probability": 0.000247380297,
+ "bob_logit": 19.7254486,
+ "bob_probability": 6.38595666e-05,
+ "generated": " \"",
+ "generated_token_id": 623,
+ "kl_from_base": 0,
+ "latency_ms": 598.812609,
+ "max_abs_logit_difference_from_base": 0,
+ "scale": 0
+ },
+ {
+ "alice_logit": 21.079689,
+ "alice_probability": 0.000247380297,
+ "bob_logit": 19.7254486,
+ "bob_probability": 6.38595666e-05,
+ "generated": " \"",
+ "generated_token_id": 623,
+ "kl_from_base": 0,
+ "latency_ms": 511.115942,
+ "max_abs_logit_difference_from_base": 0,
+ "scale": 0
+ },
+ {
+ "alice_logit": 21.079689,
+ "alice_probability": 0.000247380297,
+ "bob_logit": 19.7254486,
+ "bob_probability": 6.38595666e-05,
+ "generated": " \"",
+ "generated_token_id": 623,
+ "kl_from_base": 0,
+ "latency_ms": 504.167794,
+ "max_abs_logit_difference_from_base": 0,
+ "scale": 0
+ }
+ ],
+ "prompt": "Rain tapped the observatory windows while the old astronomer adjusted the brass lens. Guest:"
+ },
+ "prompt": "The silver key currently belongs to",
+ "prompt_tokens": 7,
+ "runner": {
+ "causal_command": [
+ "${TEMP_DIR}/planner-cache-run/llama_cpp_causal_runner",
+ "--model",
+ "${GEMMA_MODEL}",
+ "--direction",
+ "${TEMP_DIR}/planner-cache-run/gemma-direction.f32",
+ "--prompt",
+ "The silver key currently belongs to",
+ "--scales",
+ "0.0,1.999999761581421,-5.000000476837158,0.0,0.0,0.0,0.0,0.0,0.0",
+ "--layer",
+ "41",
+ "--alice-token",
+ "32858",
+ "--bob-token",
+ "15943",
+ "--gpu-layers",
+ "12"
+ ],
+ "compilation": {
+ "binary_sha256": "ac1683efac4abd95c583633ac77490abbaea860814f745b05931dc1c762e25b7",
+ "command": [
+ "c++",
+ "-std=c++17",
+ "-O2",
+ "${REPOSITORY_ROOT}/benchmarks/llama_cpp_causal_runner.cpp",
+ "-I${LOCAL_PATH}",
+ "-I${LOCAL_PATH}",
+ "${LOCAL_PATH}",
+ "${LOCAL_PATH}",
+ "-Wl,-rpath,${LOCAL_PATH}",
+ "-o",
+ "${TEMP_DIR}/planner-cache-run/llama_cpp_causal_runner"
+ ],
+ "compile_seconds": 1.2689489720032725,
+ "source_sha256": "2beaeaf65c50304ec4dca9321eb2fe542a3f3edbe9084fa293f518b3c1cd6e8f"
+ },
+ "natural_command": [
+ "${TEMP_DIR}/planner-cache-run/llama_cpp_causal_runner",
+ "--model",
+ "${GEMMA_MODEL}",
+ "--direction",
+ "${TEMP_DIR}/planner-cache-run/gemma-direction.f32",
+ "--prompt",
+ "Rain tapped the observatory windows while the old astronomer adjusted the brass lens. Guest:",
+ "--scales",
+ "0.0,0.0,0.0",
+ "--layer",
+ "41",
+ "--alice-token",
+ "32858",
+ "--bob-token",
+ "15943",
+ "--gpu-layers",
+ "12"
+ ]
+ },
+ "server_verification": {
+ "alice": {
+ "command": [
+ "${LOCAL_PATH}",
+ "--model",
+ "${GEMMA_MODEL}",
+ "--host",
+ "127.0.0.1",
+ "--port",
+ "46083",
+ "--ctx-size",
+ "512",
+ "--parallel",
+ "1",
+ "--threads",
+ "8",
+ "--threads-batch",
+ "8",
+ "--gpu-layers",
+ "12",
+ "--no-warmup",
+ "--no-context-shift",
+ "--log-disable",
+ "--control-vector-scaled",
+ "${TEMP_DIR}/planner-cache-run/gemma-direction.gguf:1.999999761581421",
+ "--control-vector-layer-range",
+ "41",
+ "41"
+ ],
+ "content": " Alice",
+ "elapsed_seconds": 3.6577627030019357,
+ "prompt_ms": 432.977,
+ "prompt_tokens": 7
+ },
+ "bob": {
+ "command": [
+ "${LOCAL_PATH}",
+ "--model",
+ "${GEMMA_MODEL}",
+ "--host",
+ "127.0.0.1",
+ "--port",
+ "34955",
+ "--ctx-size",
+ "512",
+ "--parallel",
+ "1",
+ "--threads",
+ "8",
+ "--threads-batch",
+ "8",
+ "--gpu-layers",
+ "12",
+ "--no-warmup",
+ "--no-context-shift",
+ "--log-disable",
+ "--control-vector-scaled",
+ "${TEMP_DIR}/planner-cache-run/gemma-direction.gguf:-5.000000476837158",
+ "--control-vector-layer-range",
+ "41",
+ "41"
+ ],
+ "content": " Bob",
+ "elapsed_seconds": 3.2694168429989077,
+ "prompt_ms": 418.441,
+ "prompt_tokens": 7
+ }
+ },
+ "status": "passed"
+}
diff --git a/artifacts/gemma4-e4b-q8-llama.ltl b/artifacts/gemma4-e4b-q8-llama.ltl
new file mode 100644
index 0000000000000000000000000000000000000000..f3786b24681601495337f3bab91d1eae5fee9e67
--- /dev/null
+++ b/artifacts/gemma4-e4b-q8-llama.ltl
@@ -0,0 +1 @@
+{"format":"planner-cache-ltl-v1","payload":{"adapter_class":"ltl","canonical_protocol":"pcm-canonical-p-v1","control":"direct_adaptive_logit_bias","format":"planner-cache-ltl-v1","logit_margin":0.01,"model_architecture":"gemma4","model_id":"Gemma-4-E4B-Uncensored-HauhauCS-Aggressive","model_sha256":"a4c4177f9fd7e3f56522675afb742f079a53f9226195b7db5e9888c872f053da","parameter_count":0,"runtime":"llama.cpp","runtime_version":"version: 10276 (6ea215d17)\nbuilt with GNU 16.1.1 for Linux x86_64","support_level":"lexical/output","tokenizer_bundle_sha256":"b3033e12af0ed503d8b80390c79d02d6bd9bc372e93e377cc1dd6514b7cd21d6"},"payload_sha256":"945c6d3c7f3a3668653f21733ed2b2c901c70efd314696114b70d9cf6c8c4dfa"}
diff --git a/artifacts/personality-proof.ppkg b/artifacts/personality-proof.ppkg
new file mode 100644
index 0000000000000000000000000000000000000000..e3bd46ae2606f047af624e943a712e37639ce5fa
Binary files /dev/null and b/artifacts/personality-proof.ppkg differ
diff --git a/artifacts/phase-b-factorized-representation.json b/artifacts/phase-b-factorized-representation.json
new file mode 100644
index 0000000000000000000000000000000000000000..838e715fc5978a181e52f38d94cf925af384d1c8
--- /dev/null
+++ b/artifacts/phase-b-factorized-representation.json
@@ -0,0 +1,20 @@
+{
+ "representation": {
+ "layout": {"entity": 128, "relation": 128, "value": 128, "metadata": 128, "projected_slot_width": 512},
+ "training_loss": [7.440117835998535, 0.04114125296473503],
+ "train_combinations": 2073,
+ "held_out_combinations_total": 519,
+ "held_out_combinations_probed": 256,
+ "p_only_state_recovery": 0.97265625,
+ "permutation_stability": 1.0,
+ "permutations_per_combination": 8,
+ "canonical_decode_accuracy": {"entity": 1.0, "relation": 1.0, "value": 1.0, "metadata": 1.0},
+ "hard_negative_accuracy": {"wrong_value": 1.0, "wrong_entity": 0.97265625, "historical": 1.0}
+ },
+ "pythia": {
+ "shared": {"steps": 128, "held_out_combinations": 24, "hard_negative_slots_per_state": 3, "slot_order_randomized": true, "source_tokens_in_recent_kv": 0, "held_out_rp": true},
+ "final_cross_attention": {"training_loss": [10.956916809082031, 4.057961463928223], "held_out_composition_accuracy": 0.1666666716337204, "rp_language_loss_p_disabled": 5.733096599578857, "rp_language_loss_p_enabled": 11.490976333618164, "rp_kl_with_p_to_base": 7.918457984924316},
+ "upper_4_with_preservation": {"training_loss": [10.956916809082031, 4.487305641174316], "held_out_composition_accuracy": 0.0416666679084301, "rp_language_loss_p_disabled": 5.733096599578857, "rp_language_loss_p_enabled": 5.834649562835693, "rp_kl_with_p_to_base": 0.015018677338957787}
+ },
+ "diagnosis": "P-only representation passes; exact Pythia consumption/generalization remains the blocking interface."
+}
diff --git a/artifacts/phase-b-personality-package.json b/artifacts/phase-b-personality-package.json
new file mode 100644
index 0000000000000000000000000000000000000000..fa8485791ea8156371d18e1d6a8782e1e33bdae3
--- /dev/null
+++ b/artifacts/phase-b-personality-package.json
@@ -0,0 +1,369 @@
+{
+ "base_weights_modified": false,
+ "canonical_representation_probe": {
+ "canonical_decode_accuracy": {
+ "entity": 1.0,
+ "metadata": 1.0,
+ "relation": 1.0,
+ "value": 1.0
+ },
+ "hard_negative_accuracy": {
+ "historical": 1.0,
+ "wrong_entity": 0.97265625,
+ "wrong_value": 1.0
+ },
+ "held_out_combinations": 256,
+ "p_only_state_recovery": 0.97265625,
+ "permutation_stability": 1.0,
+ "permutations_per_combination": 8,
+ "slot_width": 512,
+ "total_held_out_combinations": 519,
+ "train_combinations": 2073,
+ "training_loss_first": 7.440117835998535,
+ "training_loss_last": 0.04114125296473503
+ },
+ "completion_failures": [],
+ "completion_gate": "passed",
+ "cuda_translate": {
+ "active_memory": {
+ "context_creative": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 990,
+ "package_disk_bytes": 45056
+ },
+ "context_technical": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 996,
+ "package_disk_bytes": 45056
+ },
+ "irrelevant": {
+ "canonical_store_bytes": 0,
+ "loaded_entries": 0,
+ "logical_disk_bytes_read": 189,
+ "package_disk_bytes": 45056
+ },
+ "low_confidence": {
+ "canonical_store_bytes": 0,
+ "loaded_entries": 0,
+ "logical_disk_bytes_read": 0,
+ "package_disk_bytes": 45056
+ },
+ "package_a": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 787,
+ "package_disk_bytes": 45056
+ },
+ "package_b": {
+ "canonical_store_bytes": 1077,
+ "loaded_entries": 1,
+ "logical_disk_bytes_read": 783,
+ "package_disk_bytes": 45056
+ }
+ },
+ "base_parameters_with_grad": 0,
+ "candidate_accuracy": {
+ "context_creative_bob": 1.0,
+ "context_technical_alice": 1.0,
+ "irrelevant_matches_base": 1.0,
+ "package_a_alice": 1.0,
+ "package_b_bob": 1.0
+ },
+ "causal": {
+ "frozen_base": {
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08
+ },
+ "p_cache_only": {
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08
+ },
+ "p_cache_plus_p_package": {
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441
+ },
+ "p_package_a_relevant": {
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441
+ },
+ "p_package_b_relevant": {
+ "alice_logit": 14.0,
+ "alice_probability": 8.776464277548968e-11,
+ "bob_logit": 37.15625,
+ "bob_probability": 0.9998875856399536,
+ "gate": 0.9605370163917542,
+ "generated": " Bob",
+ "kl_from_base": 9.265372276306152
+ },
+ "p_package_context_creative": {
+ "alice_logit": 14.0,
+ "alice_probability": 8.776464277548968e-11,
+ "bob_logit": 37.15625,
+ "bob_probability": 0.9998875856399536,
+ "gate": 0.9605370163917542,
+ "generated": " Bob",
+ "kl_from_base": 9.265372276306152
+ },
+ "p_package_context_technical": {
+ "alice_logit": 40.03125,
+ "alice_probability": 1.0,
+ "bob_logit": 13.859375,
+ "bob_probability": 4.302284223323127e-12,
+ "gate": 0.9909488558769226,
+ "generated": " Alice",
+ "kl_from_base": 8.359650611877441
+ },
+ "p_package_contradictory_low_confidence": {
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08
+ },
+ "p_package_irrelevant": {
+ "alice_logit": 7.02734375,
+ "alice_probability": 0.00023412609880324453,
+ "bob_logit": 6.12109375,
+ "bob_probability": 9.459550346946344e-05,
+ "gate": 0.0,
+ "generated": " able",
+ "kl_from_base": -1.646096947638398e-08
+ }
+ },
+ "extra_prompt_tokens": 0,
+ "full_package_uploaded_to_cuda": false,
+ "inactive_package_entries": 5,
+ "inactive_vram_delta_bytes": 0,
+ "latency_seconds": {
+ "frozen_base": 0.0175060088004102,
+ "p_cache_only": 0.019222700600221288,
+ "p_cache_plus_p_package": 0.01935694159983541,
+ "p_package_a_relevant": 0.019251526400330475,
+ "p_package_b_relevant": 0.019145489999937128,
+ "p_package_context_creative": 0.019227325199608458,
+ "p_package_context_technical": 0.01928819700042368,
+ "p_package_contradictory_low_confidence": 0.0174450127997261,
+ "p_package_irrelevant": 0.017540659599762875
+ },
+ "natural_interaction": {
+ "frozen_base": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_cache_only": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_cache_plus_p_package": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_contradictory_low_confidence": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_irrelevant": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ },
+ "p_package_relevant": {
+ "kl_from_base": 4.2250388077036405e-08,
+ "loss": 5.47836971282959,
+ "samples": [
+ " \"",
+ "\n",
+ " \""
+ ]
+ }
+ },
+ "relevant_personality_chat": {
+ "base_target_loss": 8.359650611877441,
+ "generated": " Alice",
+ "package_target_loss": -0.0,
+ "target": "Alice",
+ "target_accuracy": 1.0
+ },
+ "source_tokens_in_recent_kv": 0
+ },
+ "deterministic_serialization": {
+ "byte_identical": true,
+ "first_sha256": "faa701fb6150738937fc7b756fb5df226638eedb7599127b0d04d83899ea4bf9",
+ "second_sha256": "faa701fb6150738937fc7b756fb5df226638eedb7599127b0d04d83899ea4bf9"
+ },
+ "durability": {
+ "active_entries_after_restart": 5,
+ "checksum_valid": true,
+ "cold_load_seconds": 0.0012688639981206506,
+ "conversation_replay_required": false
+ },
+ "experiment": "phase-b-personality-package-v1",
+ "format": "pcm-personality-package-v1",
+ "growth": {
+ "100": {
+ "active_canonical_bytes": 4308,
+ "cold_checksum_validation_seconds": 0.005407546999776969,
+ "disk_size_bytes": 81920,
+ "full_package_loaded": false,
+ "inactive_vram_bytes": 0,
+ "loaded_entries": 4,
+ "logical_bytes_read": 16070,
+ "lookup_latency_seconds": 0.07422141599818133,
+ "python_ram_peak_delta_bytes": 66370
+ },
+ "1000": {
+ "active_canonical_bytes": 4308,
+ "cold_checksum_validation_seconds": 0.04806955000094604,
+ "disk_size_bytes": 421888,
+ "full_package_loaded": false,
+ "inactive_vram_bytes": 0,
+ "loaded_entries": 4,
+ "logical_bytes_read": 73991,
+ "lookup_latency_seconds": 0.39593972700095037,
+ "python_ram_peak_delta_bytes": 348920
+ },
+ "10000": {
+ "active_canonical_bytes": 4308,
+ "cold_checksum_validation_seconds": 0.47560709200115525,
+ "disk_size_bytes": 3805184,
+ "full_package_loaded": false,
+ "inactive_vram_bytes": 0,
+ "loaded_entries": 4,
+ "logical_bytes_read": 73991,
+ "lookup_latency_seconds": 0.4038081879989477,
+ "python_ram_peak_delta_bytes": 296020
+ },
+ "100000": {
+ "active_canonical_bytes": 4308,
+ "cold_checksum_validation_seconds": 4.705669751998357,
+ "disk_size_bytes": 37863424,
+ "full_package_loaded": false,
+ "inactive_vram_bytes": 0,
+ "loaded_entries": 4,
+ "logical_bytes_read": 73991,
+ "lookup_latency_seconds": 0.44106237400046666,
+ "python_ram_peak_delta_bytes": 296016
+ }
+ },
+ "lora_used": false,
+ "mechanical": {
+ "change_count": 7,
+ "connected_cross_context_score": 2.586527310885154,
+ "connectivity_outscores_narrow": true,
+ "creative_context_correct": true,
+ "entry_count": 6,
+ "explicit_correction_promoted": true,
+ "irrelevant_loaded_entries": 0,
+ "narrow_context_score": 1.8000000000000003,
+ "old_conclusion_status": "superseded",
+ "one_event_promoted": false,
+ "package_size_bytes": 61440,
+ "persona_entry_id": "personality_6a0ee4fede21e1b53f3a8c4e",
+ "relationship_context_correct": true,
+ "relevance_accuracy": 1.0,
+ "repeated_entry": {
+ "confidence": 0.5567010309278351,
+ "context_diversity": 3,
+ "contradicting_evidence_ids": [
+ "weak-0"
+ ],
+ "created_at": "2026-06-03T00:00:00+00:00",
+ "entry_type": "interaction_style",
+ "evidence_count": 3,
+ "extension": {},
+ "id": "personality_a4d8044a0a392a1518bf7554",
+ "importance": 0.5,
+ "last_reinforced": "2026-06-03T00:00:00+00:00",
+ "relation": "response_style",
+ "relationship": null,
+ "scope": "global",
+ "source_authority": "single_observed_behavior",
+ "status": "active",
+ "strength": 0.7036804069636988,
+ "subject": "user",
+ "supporting_evidence_ids": [
+ "repeat-0",
+ "repeat-1",
+ "repeat-2"
+ ],
+ "updated_at": "2026-06-03T00:00:00+00:00",
+ "value": "concise"
+ },
+ "repeated_promoted": true,
+ "semantic_checksum": "abdbd3dbe3f8ee05fe761ee1625e3a8f8a4ab5259458c4d18210a15fe0b52309",
+ "technical_context_correct": true,
+ "top_k": {
+ "1": {
+ "latency_seconds": 0.0023069649978424422,
+ "loaded_entries": 1,
+ "logical_bytes_read": 1303,
+ "target_recall": 1.0
+ },
+ "4": {
+ "latency_seconds": 0.002416219998849556,
+ "loaded_entries": 4,
+ "logical_bytes_read": 3451,
+ "target_recall": 1.0
+ },
+ "8": {
+ "latency_seconds": 0.0025850960009847768,
+ "loaded_entries": 4,
+ "logical_bytes_read": 3451,
+ "target_recall": 1.0
+ }
+ },
+ "unsupported_model_claim_promoted": false
+ },
+ "package_file_sha256": "faa701fb6150738937fc7b756fb5df226638eedb7599127b0d04d83899ea4bf9",
+ "package_path": "artifacts/personality-proof.ppkg",
+ "phase_c_started": false,
+ "protocol": "pcm-canonical-personality-v1"
+}
diff --git a/artifacts/phase-b-split-translator.json b/artifacts/phase-b-split-translator.json
new file mode 100644
index 0000000000000000000000000000000000000000..3997567c702b63022619071561ffc29f561783da
--- /dev/null
+++ b/artifacts/phase-b-split-translator.json
@@ -0,0 +1,910 @@
+{
+ "canonical_p_probe": {
+ "canonical_decode_accuracy": {
+ "entity": 1.0,
+ "metadata": 1.0,
+ "relation": 1.0,
+ "value": 1.0
+ },
+ "hard_negative_accuracy": {
+ "historical": 1.0,
+ "wrong_entity": 0.97265625,
+ "wrong_value": 1.0
+ },
+ "held_out_combinations": 256,
+ "p_only_state_recovery": 0.97265625,
+ "permutation_stability": 1.0,
+ "permutations_per_combination": 8,
+ "slot_width": 512,
+ "total_held_out_combinations": 519,
+ "train_combinations": 2073,
+ "training_loss_first": 7.440117835998535,
+ "training_loss_last": 0.04114125296473503
+ },
+ "completion_gate": "passed",
+ "diagnostic_variant": "final_layer",
+ "experiment": {
+ "base_frozen": true,
+ "causal_steps": 256,
+ "lora_used": false,
+ "phase_c_started": false,
+ "query_steps": 400,
+ "router_steps": 400,
+ "seed": 307,
+ "source_tokens_in_recent_kv": 0,
+ "value_steps": 400
+ },
+ "immutable_rejected_translator_baseline": {
+ "final_layer": {
+ "exact_generated_token_accuracy": 0.45000001788139343,
+ "global_20_fact_retrieval": 0.6500000357627869,
+ "p_to_model_value_accuracy": 1.0,
+ "rp_loss": 4.656857490539551,
+ "unseen_name_accuracy": 0.25,
+ "wrong_entity_gate": 0.5172024965286255
+ },
+ "upper_2_layers": {
+ "exact_generated_token_accuracy": 0.550000011920929,
+ "global_20_fact_retrieval": 0.6500000357627869,
+ "p_to_model_value_accuracy": 1.0,
+ "rp_loss": 6.086124897003174,
+ "unseen_name_accuracy": 0.0,
+ "wrong_entity_gate": 0.6912492662668228
+ },
+ "upper_4_layers": {
+ "exact_generated_token_accuracy": 0.6500000357627869,
+ "global_20_fact_retrieval": 0.699999988079071,
+ "p_to_model_value_accuracy": 1.0,
+ "rp_loss": 5.704916477203369,
+ "unseen_name_accuracy": 0.25,
+ "wrong_entity_gate": 0.7020304128527641
+ }
+ },
+ "layer_sweep": {
+ "final_layer": {
+ "ablations": {
+ "full_system_with_preservation": {
+ "full_token_accuracy": 1.0,
+ "gate_activation": 0.9820089340209961,
+ "state_candidate_accuracy": 1.0
+ },
+ "router_only": {
+ "active_vram_overhead_bytes": 11035444,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.10144930460010074,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "router_plus_translator_plus_gate": {
+ "historical_state_kl": 9.595059236744419e-05,
+ "rp_kl": -3.993045538663864e-08,
+ "rp_loss": 5.583366870880127,
+ "state_candidate_accuracy": 1.0,
+ "state_loss": 1.8655489839147776e-05,
+ "wrong_state_kl": 9.595059236744419e-05
+ },
+ "router_plus_translator_without_gate": {
+ "state_candidate_accuracy": 1.0
+ },
+ "translator_only_oracle_routing": {
+ "state_candidate_accuracy": 1.0
+ }
+ },
+ "attachment_count": 1,
+ "attachment_layers": [
+ 23
+ ],
+ "base_parameters_with_grad": 0,
+ "causal_training_loss_first_last": [
+ 9.884638786315918,
+ 5.081248218630208e-06
+ ],
+ "counterfactual": {
+ "disabled": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p1_silver_alice": {
+ "alice_logit": 40.09375,
+ "alice_probability": 1.0,
+ "bob_logit": 14.3125,
+ "bob_probability": 6.3583643558629e-12,
+ "gate": 0.9964228272438049,
+ "generated": " Alice"
+ },
+ "p2_silver_bob": {
+ "alice_logit": 13.0234375,
+ "alice_probability": 4.473376963992637e-12,
+ "bob_logit": 39.15625,
+ "bob_probability": 0.9999349117279053,
+ "gate": 0.9842994213104248,
+ "generated": " Bob"
+ },
+ "p3_gold_alice": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p4_silver_historical": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p4_silver_invalidated": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ }
+ },
+ "extra_prompt_tokens": 0,
+ "invalidated_logit_difference": 0.0,
+ "mutation_chain": {
+ "invalidated_max_logit_difference": 0.0,
+ "latest_state_accuracy": 1.0,
+ "source_tokens_in_recent_kv": 0
+ },
+ "natural_rp": {
+ "base_loss": 5.583366870880127,
+ "conditions": {
+ "base": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "invalidated": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "irrelevant": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "wrong_entity": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ }
+ },
+ "relevant_state_generation_sample": " Alice"
+ },
+ "package_parameters": 2707464,
+ "package_path": "artifacts/pythia-1.4b-split-final_layer.translate",
+ "package_roundtrip_max_difference": 0.0,
+ "query_projector": {
+ "byte_surface_anchor_approach": {
+ "entity_accuracy": 1.0,
+ "entity_cosine": 1.0,
+ "metadata_accuracy": 1.0,
+ "oracle_slot_assignments": 0,
+ "relation_accuracy": 1.0,
+ "tokenizer_independent": true
+ },
+ "frozen_lexical_anchor_approach": {
+ "entity_accuracy": 0.78125,
+ "entity_cosine": 0.7632350921630859
+ },
+ "heldout_names": 64,
+ "hidden_to_byte_reconstruction_ablation": {
+ "entity_accuracy": 0.125,
+ "entity_cosine": 0.2643434405326843,
+ "metadata_accuracy": 1.0,
+ "relation_accuracy": 1.0
+ },
+ "loss_first_last": [
+ 6.364231586456299,
+ 0.23614102602005005
+ ],
+ "training_names": 256
+ },
+ "router": {
+ "acceptance_threshold": 5.495710372924805,
+ "calibration_balanced_accuracy": 1.0,
+ "hard_negative_metrics": {
+ "historical_false_positive_rate": 0.0,
+ "invalidated_false_positive_rate": 0.0,
+ "irrelevant_false_positive_rate": 0.0,
+ "top1_accuracy": 1.0,
+ "wrong_entity_false_positive_rate": 0.0,
+ "wrong_relation_false_positive_rate": 0.0
+ },
+ "loss_first_last": [
+ 7.388868808746338,
+ 1.7992397546768188
+ ],
+ "model_hidden_dimensions": 0,
+ "scaling": {
+ "128": {
+ "active_vram_overhead_bytes": 11035444,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.10144930460010074,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "20": {
+ "active_vram_overhead_bytes": 10861996,
+ "hidden_only_top1_accuracy": 0.20000000298023224,
+ "latency_seconds": 0.08669143320003059,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "256": {
+ "active_vram_overhead_bytes": 11241012,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.11810561999955098,
+ "mrr": 0.9505556225776672,
+ "oracle_query_mrr": 0.9505556225776672,
+ "oracle_query_top1_accuracy": 0.949999988079071,
+ "oracle_query_top4_recall": 0.949999988079071,
+ "state_generation_accuracy": 0.949999988079071,
+ "top1_accuracy": 0.949999988079071,
+ "top2_recall": 0.949999988079071,
+ "top4_recall": 0.949999988079071
+ },
+ "4": {
+ "active_vram_overhead_bytes": 10836300,
+ "hidden_only_top1_accuracy": 0.75,
+ "latency_seconds": 0.02663025279980502,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "512": {
+ "active_vram_overhead_bytes": 11652148,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.16455349219977505,
+ "mrr": 0.8524776697158813,
+ "oracle_query_mrr": 0.8524776697158813,
+ "oracle_query_top1_accuracy": 0.8500000238418579,
+ "oracle_query_top4_recall": 0.8500000238418579,
+ "state_generation_accuracy": 0.8500000238418579,
+ "top1_accuracy": 0.8500000238418579,
+ "top2_recall": 0.8500000238418579,
+ "top4_recall": 0.8500000238418579
+ },
+ "64": {
+ "active_vram_overhead_bytes": 10932660,
+ "hidden_only_top1_accuracy": 0.05000000074505806,
+ "latency_seconds": 0.09232893199950923,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ }
+ }
+ },
+ "router_parameters": 5,
+ "router_path": "artifacts/canonical-final_layer.router",
+ "router_roundtrip_max_difference": 0.0,
+ "source_tokens_in_recent_kv": 0,
+ "value_translator": {
+ "loss_first_last": [
+ 4.713489055633545,
+ 0.00027370601310394704
+ ],
+ "oracle_selected_metrics": {
+ "accuracy": 1.0,
+ "cosine": 0.9998512268066406
+ }
+ }
+ },
+ "upper_2_layers": {
+ "ablations": {
+ "full_system_with_preservation": {
+ "full_token_accuracy": 1.0,
+ "gate_activation": 0.8225321769714355,
+ "state_candidate_accuracy": 1.0
+ },
+ "router_only": {
+ "active_vram_overhead_bytes": 11035444,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.10514904640003805,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "router_plus_translator_plus_gate": {
+ "historical_state_kl": 9.595059236744419e-05,
+ "rp_kl": -3.993045538663864e-08,
+ "rp_loss": 5.583366870880127,
+ "state_candidate_accuracy": 1.0,
+ "state_loss": 5.1855395213351585e-06,
+ "wrong_state_kl": 9.595059236744419e-05
+ },
+ "router_plus_translator_without_gate": {
+ "state_candidate_accuracy": 1.0
+ },
+ "translator_only_oracle_routing": {
+ "state_candidate_accuracy": 1.0
+ }
+ },
+ "attachment_count": 2,
+ "attachment_layers": [
+ 22,
+ 23
+ ],
+ "base_parameters_with_grad": 0,
+ "causal_training_loss_first_last": [
+ 9.555724143981934,
+ 1.2963974995727767e-06
+ ],
+ "counterfactual": {
+ "disabled": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p1_silver_alice": {
+ "alice_logit": 58.59375,
+ "alice_probability": 1.0,
+ "bob_logit": 18.5,
+ "bob_probability": 3.868170754671019e-18,
+ "gate": 0.941794216632843,
+ "generated": " Alice"
+ },
+ "p2_silver_bob": {
+ "alice_logit": 16.078125,
+ "alice_probability": 6.54239844304198e-17,
+ "bob_logit": 53.34375,
+ "bob_probability": 0.9999822378158569,
+ "gate": 0.8498331308364868,
+ "generated": " Bob"
+ },
+ "p3_gold_alice": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p4_silver_historical": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p4_silver_invalidated": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ }
+ },
+ "extra_prompt_tokens": 0,
+ "invalidated_logit_difference": 0.0,
+ "mutation_chain": {
+ "invalidated_max_logit_difference": 0.0,
+ "latest_state_accuracy": 1.0,
+ "source_tokens_in_recent_kv": 0
+ },
+ "natural_rp": {
+ "base_loss": 5.583366870880127,
+ "conditions": {
+ "base": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "invalidated": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "irrelevant": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "wrong_entity": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ }
+ },
+ "relevant_state_generation_sample": " Alice"
+ },
+ "package_parameters": 2707464,
+ "package_path": "artifacts/pythia-1.4b-split-upper_2_layers.translate",
+ "package_roundtrip_max_difference": 0.0,
+ "query_projector": {
+ "byte_surface_anchor_approach": {
+ "entity_accuracy": 1.0,
+ "entity_cosine": 1.0,
+ "metadata_accuracy": 1.0,
+ "oracle_slot_assignments": 0,
+ "relation_accuracy": 1.0,
+ "tokenizer_independent": true
+ },
+ "frozen_lexical_anchor_approach": {
+ "entity_accuracy": 0.78125,
+ "entity_cosine": 0.7632350921630859
+ },
+ "heldout_names": 64,
+ "hidden_to_byte_reconstruction_ablation": {
+ "entity_accuracy": 0.046875,
+ "entity_cosine": 0.281654953956604,
+ "metadata_accuracy": 1.0,
+ "relation_accuracy": 1.0
+ },
+ "loss_first_last": [
+ 6.3322367668151855,
+ 0.24513374269008636
+ ],
+ "training_names": 256
+ },
+ "router": {
+ "acceptance_threshold": 5.495710372924805,
+ "calibration_balanced_accuracy": 1.0,
+ "hard_negative_metrics": {
+ "historical_false_positive_rate": 0.0,
+ "invalidated_false_positive_rate": 0.0,
+ "irrelevant_false_positive_rate": 0.0,
+ "top1_accuracy": 1.0,
+ "wrong_entity_false_positive_rate": 0.0,
+ "wrong_relation_false_positive_rate": 0.0
+ },
+ "loss_first_last": [
+ 7.388868808746338,
+ 1.7992397546768188
+ ],
+ "model_hidden_dimensions": 0,
+ "scaling": {
+ "128": {
+ "active_vram_overhead_bytes": 11035444,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.10514904640003805,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "20": {
+ "active_vram_overhead_bytes": 10861996,
+ "hidden_only_top1_accuracy": 0.10000000149011612,
+ "latency_seconds": 0.08956059680058388,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "256": {
+ "active_vram_overhead_bytes": 11241012,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.12291700120040332,
+ "mrr": 0.9505556225776672,
+ "oracle_query_mrr": 0.9505556225776672,
+ "oracle_query_top1_accuracy": 0.949999988079071,
+ "oracle_query_top4_recall": 0.949999988079071,
+ "state_generation_accuracy": 0.949999988079071,
+ "top1_accuracy": 0.949999988079071,
+ "top2_recall": 0.949999988079071,
+ "top4_recall": 0.949999988079071
+ },
+ "4": {
+ "active_vram_overhead_bytes": 10836300,
+ "hidden_only_top1_accuracy": 0.5,
+ "latency_seconds": 0.028453471799730325,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "512": {
+ "active_vram_overhead_bytes": 11652148,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.17295584960011184,
+ "mrr": 0.8524776697158813,
+ "oracle_query_mrr": 0.8524776697158813,
+ "oracle_query_top1_accuracy": 0.8500000238418579,
+ "oracle_query_top4_recall": 0.8500000238418579,
+ "state_generation_accuracy": 0.8500000238418579,
+ "top1_accuracy": 0.8500000238418579,
+ "top2_recall": 0.8500000238418579,
+ "top4_recall": 0.8500000238418579
+ },
+ "64": {
+ "active_vram_overhead_bytes": 10932660,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.09586576699948637,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ }
+ }
+ },
+ "router_parameters": 5,
+ "router_path": "artifacts/canonical-upper_2_layers.router",
+ "router_roundtrip_max_difference": 0.0,
+ "source_tokens_in_recent_kv": 0,
+ "value_translator": {
+ "loss_first_last": [
+ 4.713489055633545,
+ 0.00027370601310394704
+ ],
+ "oracle_selected_metrics": {
+ "accuracy": 1.0,
+ "cosine": 0.9998512268066406
+ }
+ }
+ },
+ "upper_4_layers": {
+ "ablations": {
+ "full_system_with_preservation": {
+ "full_token_accuracy": 1.0,
+ "gate_activation": 0.6373323798179626,
+ "state_candidate_accuracy": 1.0
+ },
+ "router_only": {
+ "active_vram_overhead_bytes": 11035444,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.11078032340010395,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "router_plus_translator_plus_gate": {
+ "historical_state_kl": 9.595059236744419e-05,
+ "rp_kl": -3.993045538663864e-08,
+ "rp_loss": 5.583366870880127,
+ "state_candidate_accuracy": 1.0,
+ "state_loss": 1.4752099559700582e-06,
+ "wrong_state_kl": 9.595059236744419e-05
+ },
+ "router_plus_translator_without_gate": {
+ "state_candidate_accuracy": 1.0
+ },
+ "translator_only_oracle_routing": {
+ "state_candidate_accuracy": 1.0
+ }
+ },
+ "attachment_count": 4,
+ "attachment_layers": [
+ 20,
+ 21,
+ 22,
+ 23
+ ],
+ "base_parameters_with_grad": 0,
+ "causal_training_loss_first_last": [
+ 8.895807266235352,
+ 6.407490786841663e-07
+ ],
+ "counterfactual": {
+ "disabled": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p1_silver_alice": {
+ "alice_logit": 72.5625,
+ "alice_probability": 1.0,
+ "bob_logit": 20.375,
+ "bob_probability": 2.1639972398594e-23,
+ "gate": 0.8317975997924805,
+ "generated": " Alice"
+ },
+ "p2_silver_bob": {
+ "alice_logit": 17.796875,
+ "alice_probability": 7.510040636028562e-22,
+ "bob_logit": 66.4375,
+ "bob_probability": 0.9999938011169434,
+ "gate": 0.6602288484573364,
+ "generated": " Bob"
+ },
+ "p3_gold_alice": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p4_silver_historical": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ },
+ "p4_silver_invalidated": {
+ "alice_logit": 5.515625,
+ "alice_probability": 3.538730743457563e-05,
+ "bob_logit": 6.2734375,
+ "bob_probability": 7.55025030230172e-05,
+ "gate": 0.0,
+ "generated": " a"
+ }
+ },
+ "extra_prompt_tokens": 0,
+ "invalidated_logit_difference": 0.0,
+ "mutation_chain": {
+ "invalidated_max_logit_difference": 0.0,
+ "latest_state_accuracy": 1.0,
+ "source_tokens_in_recent_kv": 0
+ },
+ "natural_rp": {
+ "base_loss": 5.583366870880127,
+ "conditions": {
+ "base": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "invalidated": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "irrelevant": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ },
+ "wrong_entity": {
+ "kl": -3.993045538663864e-08,
+ "loss": 5.583366870880127,
+ "samples": [
+ "\n\n\"I'm not",
+ "\n\n\"I'm going"
+ ]
+ }
+ },
+ "relevant_state_generation_sample": " Alice"
+ },
+ "package_parameters": 2707464,
+ "package_path": "artifacts/pythia-1.4b-split-upper_4_layers.translate",
+ "package_roundtrip_max_difference": 0.0,
+ "query_projector": {
+ "byte_surface_anchor_approach": {
+ "entity_accuracy": 1.0,
+ "entity_cosine": 1.0,
+ "metadata_accuracy": 1.0,
+ "oracle_slot_assignments": 0,
+ "relation_accuracy": 1.0,
+ "tokenizer_independent": true
+ },
+ "frozen_lexical_anchor_approach": {
+ "entity_accuracy": 0.78125,
+ "entity_cosine": 0.7632350921630859
+ },
+ "heldout_names": 64,
+ "hidden_to_byte_reconstruction_ablation": {
+ "entity_accuracy": 0.078125,
+ "entity_cosine": 0.28708386421203613,
+ "metadata_accuracy": 1.0,
+ "relation_accuracy": 1.0
+ },
+ "loss_first_last": [
+ 6.471477031707764,
+ 0.2573241889476776
+ ],
+ "training_names": 256
+ },
+ "router": {
+ "acceptance_threshold": 5.495710372924805,
+ "calibration_balanced_accuracy": 1.0,
+ "hard_negative_metrics": {
+ "historical_false_positive_rate": 0.0,
+ "invalidated_false_positive_rate": 0.0,
+ "irrelevant_false_positive_rate": 0.0,
+ "top1_accuracy": 1.0,
+ "wrong_entity_false_positive_rate": 0.0,
+ "wrong_relation_false_positive_rate": 0.0
+ },
+ "loss_first_last": [
+ 7.388868808746338,
+ 1.7992397546768188
+ ],
+ "model_hidden_dimensions": 0,
+ "scaling": {
+ "128": {
+ "active_vram_overhead_bytes": 11035444,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.11078032340010395,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "20": {
+ "active_vram_overhead_bytes": 10861996,
+ "hidden_only_top1_accuracy": 0.10000000149011612,
+ "latency_seconds": 0.09917421240024851,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "256": {
+ "active_vram_overhead_bytes": 11241012,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.1296995702003187,
+ "mrr": 0.9505556225776672,
+ "oracle_query_mrr": 0.9505556225776672,
+ "oracle_query_top1_accuracy": 0.949999988079071,
+ "oracle_query_top4_recall": 0.949999988079071,
+ "state_generation_accuracy": 0.949999988079071,
+ "top1_accuracy": 0.949999988079071,
+ "top2_recall": 0.949999988079071,
+ "top4_recall": 0.949999988079071
+ },
+ "4": {
+ "active_vram_overhead_bytes": 10836300,
+ "hidden_only_top1_accuracy": 0.25,
+ "latency_seconds": 0.031837903400446524,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ },
+ "512": {
+ "active_vram_overhead_bytes": 11652148,
+ "hidden_only_top1_accuracy": 0.0,
+ "latency_seconds": 0.18178108020001674,
+ "mrr": 0.8524776697158813,
+ "oracle_query_mrr": 0.8524776697158813,
+ "oracle_query_top1_accuracy": 0.8500000238418579,
+ "oracle_query_top4_recall": 0.8500000238418579,
+ "state_generation_accuracy": 0.8500000238418579,
+ "top1_accuracy": 0.8500000238418579,
+ "top2_recall": 0.8500000238418579,
+ "top4_recall": 0.8500000238418579
+ },
+ "64": {
+ "active_vram_overhead_bytes": 10932660,
+ "hidden_only_top1_accuracy": 0.05000000074505806,
+ "latency_seconds": 0.10435858960045152,
+ "mrr": 1.0,
+ "oracle_query_mrr": 1.0,
+ "oracle_query_top1_accuracy": 1.0,
+ "oracle_query_top4_recall": 1.0,
+ "state_generation_accuracy": 1.0,
+ "top1_accuracy": 1.0,
+ "top2_recall": 1.0,
+ "top4_recall": 1.0
+ }
+ }
+ },
+ "router_parameters": 5,
+ "router_path": "artifacts/canonical-upper_4_layers.router",
+ "router_roundtrip_max_difference": 0.0,
+ "source_tokens_in_recent_kv": 0,
+ "value_translator": {
+ "loss_first_last": [
+ 4.713489055633545,
+ 0.00027370601310394704
+ ],
+ "oracle_selected_metrics": {
+ "accuracy": 1.0,
+ "cosine": 0.9998512268066406
+ }
+ }
+ }
+ },
+ "selected_variant": "final_layer"
+}
diff --git a/artifacts/post-turn-memory-review-acceptance.json b/artifacts/post-turn-memory-review-acceptance.json
new file mode 100644
index 0000000000000000000000000000000000000000..73040a3e7433cc2d03e9eb207236f3e8afed07bc
--- /dev/null
+++ b/artifacts/post-turn-memory-review-acceptance.json
@@ -0,0 +1,58 @@
+{
+ "format": "planner-cache-post-turn-memory-review-acceptance-v1",
+ "date": "2026-08-23",
+ "gemma": {
+ "runtime": "Gemma4 E4B Q8 through llama.cpp",
+ "reviewer": "same frozen Gemma runtime without LTL",
+ "session_id": "benchmark-session-gemma",
+ "turns": [
+ {
+ "user": "*I leave the brass key inside the kitchen drawer.*",
+ "operation": "CREATE",
+ "entity": "brass key",
+ "relation": "location",
+ "value": "kitchen drawer",
+ "source": "rp_action",
+ "confidence": 1.0,
+ "review_latency_seconds": 43.14148591599951
+ },
+ {
+ "user": "*I take the key and put it into my coat pocket.*",
+ "operation": "MODIFY",
+ "entity": "brass key",
+ "relation": "location",
+ "value": "coat pocket",
+ "source": "rp_action",
+ "confidence": 1.0,
+ "review_latency_seconds": 48.18768195499433
+ }
+ ],
+ "final_active_state_count": 1,
+ "final_value": "coat pocket",
+ "stale_kitchen_drawer_active": false
+ },
+ "pythia": {
+ "runtime": "frozen Pythia-1.4B with TTL",
+ "reviewer": "frozen Gemma llama.cpp CPU structured reviewer without TTL or LTL",
+ "session_id": "benchmark-session-pythia",
+ "turn": {
+ "user": "*I leave the brass key inside the kitchen drawer.*",
+ "operation": "CREATE",
+ "entity": "brass key",
+ "relation": "location",
+ "value": "kitchen drawer",
+ "source": "rp_action",
+ "confidence": 1.0,
+ "review_latency_seconds": 65.49500731100852
+ },
+ "final_active_state_count": 1
+ },
+ "assertions": {
+ "natural_rp_create": true,
+ "natural_rp_modify": true,
+ "single_current_value": true,
+ "pythia_interactive_path": true,
+ "gemma_interactive_path": true,
+ "review_ttl_ltl_disabled": true
+ }
+}
diff --git a/artifacts/ppkg-100k-profile.json b/artifacts/ppkg-100k-profile.json
new file mode 100644
index 0000000000000000000000000000000000000000..367a59b984c77c39bb1d59b606b822cc57493da6
--- /dev/null
+++ b/artifacts/ppkg-100k-profile.json
@@ -0,0 +1,55 @@
+{
+ "after_reused_connection_per_query": {
+ "canonical_conversion_seconds": 0.0007130039994081017,
+ "checksum_work_seconds": 0.0,
+ "db_open_seconds": 0.0,
+ "model": "one validated open, then bounded queries on one reused connection",
+ "routing_header_query_seconds": 0.25507766800001264,
+ "row_hydration_seconds": 0.00020283000048948452,
+ "total_seconds": 0.2559935019999102
+ },
+ "before_legacy_per_query": {
+ "canonical_conversion_seconds": 0.0007130039994081017,
+ "checksum_work_seconds": 0.8436194509995403,
+ "db_open_seconds": 0.0003522689985402394,
+ "model": "validated open and whole-package checksum on every query",
+ "routing_header_query_seconds": 0.25507766800001264,
+ "row_hydration_seconds": 0.00020283000048948452,
+ "total_seconds": 1.0999652219979907
+ },
+ "build_and_initial_checkpoint_seconds": 2.8209503369980666,
+ "cold_integrity_boundary": {
+ "checksum_calls": 1,
+ "explicit_checksum_verify_seconds": 0.8436194509995403,
+ "unverified_db_open_seconds": 0.0003522689985402394
+ },
+ "connection": {
+ "connection_identity_stable": true,
+ "per_query_open_seconds": 0.0,
+ "reused": true
+ },
+ "entries": 100000,
+ "evidence_push": {
+ "checkpoint_checksum_calls": 1,
+ "checkpoint_seconds": 0.8296411900009844,
+ "legacy_modeled_push_seconds": 0.831370665004215,
+ "normal_push_checksum_calls": 0,
+ "normal_push_seconds": 0.0017294750032306183
+ },
+ "experiment": "ppkg-100k-integrity-boundary-profile-v1",
+ "modeled_speedup": 4.296848214523728,
+ "normal_query_checksum_calls": 0,
+ "package_size_bytes": 37863424,
+ "repeats": 7,
+ "selected": {
+ "entries_hydrated": 4,
+ "entry_ids": [
+ "personality-00000000",
+ "personality-00000403",
+ "personality-00000186",
+ "personality-00000124"
+ ],
+ "logical_entry_bytes": 2405
+ },
+ "sqlite_transactions_changed": false
+}
diff --git a/artifacts/pythia-1.4b-final-layer.ttl b/artifacts/pythia-1.4b-final-layer.ttl
new file mode 100644
index 0000000000000000000000000000000000000000..9776bf2b094d32bb9ddd434c6f1398759e73b593
--- /dev/null
+++ b/artifacts/pythia-1.4b-final-layer.ttl
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:72ef68d07ee27c37b90432d34d4be5c2c280ae1bcb08236e37a0e458c054d8d7
+size 10832776
diff --git a/artifacts/vram-comparison.json b/artifacts/vram-comparison.json
new file mode 100644
index 0000000000000000000000000000000000000000..ec911fcb85b4f24386cd7636831b3ba7192755ae
--- /dev/null
+++ b/artifacts/vram-comparison.json
@@ -0,0 +1,268 @@
+{
+ "completion": {
+ "conditions": 9,
+ "estimated_values": 0,
+ "failed": 0,
+ "oom_events": 0,
+ "successful": 9
+ },
+ "environment": {
+ "cuda_driver": "610.57.04",
+ "cuda_runtime": "13.0",
+ "gpu": "NVIDIA GeForce RTX 3050 Laptop GPU",
+ "gpu_total_bytes": 3950575616,
+ "platform": "Linux-7.1.8-1-MANJARO-x86_64-with-glibc2.44",
+ "python": "3.12.6",
+ "torch": "2.13.0+cu130",
+ "transformers": "5.15.1"
+ },
+ "experiment": "planner-cache-matched-vram-comparison-v1",
+ "method": {
+ "attention_working_memory": "present in every condition and distinct from retained KV cache",
+ "baseline": "CUDA allocation after loaded stack and empty_cache",
+ "combined": "retained KV and P-cache TTL path active together",
+ "kv_only": "retained KV active while P-cache and TTL injection are disabled",
+ "model_load": "one frozen model, TTL, and router reused for every condition",
+ "p_cache_only": "retained KV disabled while P-cache and TTL are active",
+ "peak": "torch.cuda reset_peak_memory_stats and synchronized measurement",
+ "warmup": "all measured mechanisms exercised once before baselines"
+ },
+ "model": {
+ "config_sha256": "6ea552aa42b7437f019ccdd30b7c9b83a32dccb5170ce31b9ddbe94d00f4671c",
+ "identifier": "pythia-1.4b",
+ "router_filename": "canonical-p-v1.router",
+ "router_sha256": "29e0728c12f8806c48998579aedcfa389fa398f909098e162710eb2a5709834e",
+ "ttl_filename": "pythia-1.4b-final-layer.ttl",
+ "ttl_sha256": "72ef68d07ee27c37b90432d34d4be5c2c280ae1bcb08236e37a0e458c054d8d7"
+ },
+ "results": [
+ {
+ "baseline_allocated_bytes": 2849172992,
+ "baseline_reserved_bytes": 2906652672,
+ "batch_size": 1,
+ "condition": "p_cache_only",
+ "failure": null,
+ "fallback": null,
+ "final_sequence_tokens": 72,
+ "generated_tokens": 8,
+ "generation": "greedy argmax",
+ "incremental_peak_allocated_bytes": 7471616,
+ "incremental_peak_reserved_bytes": 4194304,
+ "p_cache_canonical_bytes": 134464,
+ "p_cache_enabled": true,
+ "p_cache_slots": 64,
+ "peak_allocated_bytes": 2856644608,
+ "peak_reserved_bytes": 2910846976,
+ "precision": "float16 base and float32 TTL",
+ "prompt_tokens": 64,
+ "requested_generated_tokens": 8,
+ "retained_kv_cache_bytes": 0,
+ "retained_kv_enabled": false,
+ "runtime_seconds": 0.2554637499997625
+ },
+ {
+ "baseline_allocated_bytes": 2849172992,
+ "baseline_reserved_bytes": 2906652672,
+ "batch_size": 1,
+ "condition": "kv_only",
+ "failure": null,
+ "fallback": null,
+ "final_sequence_tokens": 72,
+ "generated_tokens": 8,
+ "generation": "greedy argmax",
+ "incremental_peak_allocated_bytes": 19284992,
+ "incremental_peak_reserved_bytes": 16777216,
+ "p_cache_canonical_bytes": 0,
+ "p_cache_enabled": false,
+ "p_cache_slots": 0,
+ "peak_allocated_bytes": 2868457984,
+ "peak_reserved_bytes": 2923429888,
+ "precision": "float16 base and float32 TTL",
+ "prompt_tokens": 64,
+ "requested_generated_tokens": 8,
+ "retained_kv_cache_bytes": 13959168,
+ "retained_kv_enabled": true,
+ "runtime_seconds": 0.17538804200012237
+ },
+ {
+ "baseline_allocated_bytes": 2849172992,
+ "baseline_reserved_bytes": 2906652672,
+ "batch_size": 1,
+ "condition": "p_cache_plus_kv",
+ "failure": null,
+ "fallback": null,
+ "final_sequence_tokens": 72,
+ "generated_tokens": 8,
+ "generation": "greedy argmax",
+ "incremental_peak_allocated_bytes": 19319808,
+ "incremental_peak_reserved_bytes": 16777216,
+ "p_cache_canonical_bytes": 134464,
+ "p_cache_enabled": true,
+ "p_cache_slots": 64,
+ "peak_allocated_bytes": 2868492800,
+ "peak_reserved_bytes": 2923429888,
+ "precision": "float16 base and float32 TTL",
+ "prompt_tokens": 64,
+ "requested_generated_tokens": 8,
+ "retained_kv_cache_bytes": 13959168,
+ "retained_kv_enabled": true,
+ "runtime_seconds": 0.20454967800469603
+ },
+ {
+ "baseline_allocated_bytes": 2849172992,
+ "baseline_reserved_bytes": 2906652672,
+ "batch_size": 1,
+ "condition": "p_cache_only",
+ "failure": null,
+ "fallback": null,
+ "final_sequence_tokens": 264,
+ "generated_tokens": 8,
+ "generation": "greedy argmax",
+ "incremental_peak_allocated_bytes": 28483584,
+ "incremental_peak_reserved_bytes": 35651584,
+ "p_cache_canonical_bytes": 537856,
+ "p_cache_enabled": true,
+ "p_cache_slots": 256,
+ "peak_allocated_bytes": 2877656576,
+ "peak_reserved_bytes": 2942304256,
+ "precision": "float16 base and float32 TTL",
+ "prompt_tokens": 256,
+ "requested_generated_tokens": 8,
+ "retained_kv_cache_bytes": 0,
+ "retained_kv_enabled": false,
+ "runtime_seconds": 0.7414951529935934
+ },
+ {
+ "baseline_allocated_bytes": 2849172992,
+ "baseline_reserved_bytes": 2906652672,
+ "batch_size": 1,
+ "condition": "kv_only",
+ "failure": null,
+ "fallback": null,
+ "final_sequence_tokens": 264,
+ "generated_tokens": 8,
+ "generation": "greedy argmax",
+ "incremental_peak_allocated_bytes": 81187328,
+ "incremental_peak_reserved_bytes": 83886080,
+ "p_cache_canonical_bytes": 0,
+ "p_cache_enabled": false,
+ "p_cache_slots": 0,
+ "peak_allocated_bytes": 2930360320,
+ "peak_reserved_bytes": 2990538752,
+ "precision": "float16 base and float32 TTL",
+ "prompt_tokens": 256,
+ "requested_generated_tokens": 8,
+ "retained_kv_cache_bytes": 51707904,
+ "retained_kv_enabled": true,
+ "runtime_seconds": 0.19367339000746142
+ },
+ {
+ "baseline_allocated_bytes": 2849172992,
+ "baseline_reserved_bytes": 2906652672,
+ "batch_size": 1,
+ "condition": "p_cache_plus_kv",
+ "failure": null,
+ "fallback": null,
+ "final_sequence_tokens": 264,
+ "generated_tokens": 8,
+ "generation": "greedy argmax",
+ "incremental_peak_allocated_bytes": 81323520,
+ "incremental_peak_reserved_bytes": 83886080,
+ "p_cache_canonical_bytes": 537856,
+ "p_cache_enabled": true,
+ "p_cache_slots": 256,
+ "peak_allocated_bytes": 2930496512,
+ "peak_reserved_bytes": 2990538752,
+ "precision": "float16 base and float32 TTL",
+ "prompt_tokens": 256,
+ "requested_generated_tokens": 8,
+ "retained_kv_cache_bytes": 51707904,
+ "retained_kv_enabled": true,
+ "runtime_seconds": 0.41562580699974205
+ },
+ {
+ "baseline_allocated_bytes": 2849172992,
+ "baseline_reserved_bytes": 2906652672,
+ "batch_size": 1,
+ "condition": "p_cache_only",
+ "failure": null,
+ "fallback": null,
+ "final_sequence_tokens": 1032,
+ "generated_tokens": 8,
+ "generation": "greedy argmax",
+ "incremental_peak_allocated_bytes": 108517888,
+ "incremental_peak_reserved_bytes": 174063616,
+ "p_cache_canonical_bytes": 2151424,
+ "p_cache_enabled": true,
+ "p_cache_slots": 1024,
+ "peak_allocated_bytes": 2957690880,
+ "peak_reserved_bytes": 3080716288,
+ "precision": "float16 base and float32 TTL",
+ "prompt_tokens": 1024,
+ "requested_generated_tokens": 8,
+ "retained_kv_cache_bytes": 0,
+ "retained_kv_enabled": false,
+ "runtime_seconds": 2.710468404009589
+ },
+ {
+ "baseline_allocated_bytes": 2849172992,
+ "baseline_reserved_bytes": 2906652672,
+ "batch_size": 1,
+ "condition": "kv_only",
+ "failure": null,
+ "fallback": null,
+ "final_sequence_tokens": 1032,
+ "generated_tokens": 8,
+ "generation": "greedy argmax",
+ "incremental_peak_allocated_bytes": 308559872,
+ "incremental_peak_reserved_bytes": 339738624,
+ "p_cache_canonical_bytes": 0,
+ "p_cache_enabled": false,
+ "p_cache_slots": 0,
+ "peak_allocated_bytes": 3157732864,
+ "peak_reserved_bytes": 3246391296,
+ "precision": "float16 base and float32 TTL",
+ "prompt_tokens": 1024,
+ "requested_generated_tokens": 8,
+ "retained_kv_cache_bytes": 202702848,
+ "retained_kv_enabled": true,
+ "runtime_seconds": 0.37531103400397114
+ },
+ {
+ "baseline_allocated_bytes": 2849172992,
+ "baseline_reserved_bytes": 2906652672,
+ "batch_size": 1,
+ "condition": "p_cache_plus_kv",
+ "failure": null,
+ "fallback": null,
+ "final_sequence_tokens": 1032,
+ "generated_tokens": 8,
+ "generation": "greedy argmax",
+ "incremental_peak_allocated_bytes": 309102080,
+ "incremental_peak_reserved_bytes": 358612992,
+ "p_cache_canonical_bytes": 2151424,
+ "p_cache_enabled": true,
+ "p_cache_slots": 1024,
+ "peak_allocated_bytes": 3158275072,
+ "peak_reserved_bytes": 3265265664,
+ "precision": "float16 base and float32 TTL",
+ "prompt_tokens": 1024,
+ "requested_generated_tokens": 8,
+ "retained_kv_cache_bytes": 202702848,
+ "retained_kv_enabled": true,
+ "runtime_seconds": 1.2812230650015408
+ }
+ ],
+ "shared_configuration": {
+ "batch_size": 1,
+ "generated_tokens": 8,
+ "generation": "greedy argmax",
+ "precision": "float16 base and float32 TTL",
+ "seed": 317,
+ "workload_prompt_and_slot_sizes": [
+ 64,
+ 256,
+ 1024
+ ]
+ }
+}
diff --git a/assets/EVIDENCE_MANIFEST.json b/assets/EVIDENCE_MANIFEST.json
new file mode 100644
index 0000000000000000000000000000000000000000..5598584cb4833cdf7fae7ba5a1f2643730792543
--- /dev/null
+++ b/assets/EVIDENCE_MANIFEST.json
@@ -0,0 +1,38 @@
+{
+ "generated_from": {
+ "active-system-audit.json": "6eb3e401023b5878b2bbfb9055fd921999b571f6216e370b12440b7d343c0a1d",
+ "active-system-cuda-attribution.json": "ffe0e8f32f9a1a059767607f0c5e0432b6baf2a9917dc56877f794458b1a5e1c",
+ "debug-actions-profile.json": "3ca4cc8ac002dc30bd6c33d1bdae764bfc3e43534498eecc98a5168069784bd4",
+ "gemma-native-prompt-equivalence.json": "f6823b9b5a5fd69ebf75c8aadc8efa23fe0a6eaac7a1eaa81c398a3fdb200e79",
+ "gemma4-e4b-q8-causal.json": "a15a862dbc5dde547e7124ce3c3f16d015c15b79730bc60b7e92425e471ecb0f",
+ "phase-b-factorized-representation.json": "602bf8b8b1d07b8100b13f9842790f87ae6a5229ddc2d66377130d15413a870f",
+ "phase-b-personality-package.json": "5d17cbd8cecc8398105d3d53ff05e1b7941dd0732ac3d68a761790b64799bf06",
+ "phase-b-split-translator.json": "4a8cdc0c2bc73a1fc10f974f20d88be4902331a07b92928edb089b852bc1a845",
+ "post-turn-memory-review-acceptance.json": "89cb901c1665e9fd1515ef9165580a4bd3c4d9774f23976fdc6f366ef88c6b24",
+ "ppkg-100k-profile.json": "f5ff5233632bf5bf860fefedeec560e10a3f32729f4b6a767d0b3538d9c33a05",
+ "vram-comparison.json": "1b1e266c3f6513a5708711f09879a6519ce45abdaea7ba16d04f2510f5c1fc8d"
+ },
+ "public_binary_artifacts": {
+ "canonical-p-v1.router": "29e0728c12f8806c48998579aedcfa389fa398f909098e162710eb2a5709834e",
+ "gemma4-e4b-q8-llama.ltl": "7eee07ed813796dd0d25b04b48aec73507b934479ad8902e1b657653c040781a",
+ "personality-proof.ppkg": "faa701fb6150738937fc7b756fb5df226638eedb7599127b0d04d83899ea4bf9",
+ "pythia-1.4b-final-layer.ttl": "72ef68d07ee27c37b90432d34d4be5c2c280ae1bcb08236e37a0e458c054d8d7"
+ },
+ "publication_copy": {
+ "machine_paths_normalized": true,
+ "third_party_model_or_tokenizer_payloads": false
+ },
+ "selected_configuration": {
+ "base_model": "pythia-1.4b",
+ "canonical_protocol": "pcm-canonical-p-v1",
+ "gemma_ltl_format": "planner-cache-ltl-v1",
+ "gemma_ltl_parameters": 0,
+ "gemma_model": "Gemma-4-E4B-Uncensored-HauhauCS-Aggressive",
+ "ppkg_format": "pcm-personality-package-v1",
+ "ppkg_protocol": "pcm-canonical-personality-v1",
+ "router_format": "pcm-canonical-router-v1",
+ "selected_attachment": "final_layer",
+ "ttl_format": "planner-cache-ttl-v1",
+ "ttl_parameters": 2707464
+ }
+}
diff --git a/assets/README.md b/assets/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..b83ad278e16e7488dbc23a79a74cfc2e993baf5a
--- /dev/null
+++ b/assets/README.md
@@ -0,0 +1,31 @@
+# Shared publication assets
+
+These assets are shared by the GitHub, Hugging Face, and research publication packs.
+
+| Asset | Purpose | Source |
+|---|---|---|
+| [`architecture.svg`](architecture.svg) | Repository-friendly current architecture diagram | Publication architecture specification in `generate_assets.py` |
+| [`architecture.pdf`](architecture.pdf) | Vector publication architecture figure | Generated from `architecture.svg` |
+| [`architecture.mmd`](architecture.mmd) | Compact Mermaid architecture diagram | Current architecture boundaries |
+| [`router_scaling.svg`](router_scaling.svg) | Post-audit router accuracy and MRR plot | `active-system-audit.json` |
+| [`router_scaling.csv`](router_scaling.csv) | Router plot data | `active-system-audit.json` |
+| [`ppkg_lookup.svg`](ppkg_lookup.svg) | P-package lookup latency plot | `active-system-audit.json` |
+| [`ppkg_scaling.csv`](ppkg_scaling.csv) | P-package growth and lookup table | `active-system-audit.json` |
+| [`causal_conditions.csv`](causal_conditions.csv) | Matched CUDA causal attribution table | `active-system-cuda-attribution.json` |
+| [`gemma_causal_conditions.csv`](gemma_causal_conditions.csv) | Gemma 4 Q8 GGUF causal conditions | `gemma4-e4b-q8-causal.json` |
+| [`vram_comparison.csv`](vram_comparison.csv) | Matched P-cache, retained-KV, and combined VRAM table | `vram-comparison.json` |
+| [`vram_comparison.svg`](vram_comparison.svg) | Matched incremental peak VRAM plot | `vram-comparison.json` |
+| [`VRAM_COMPARISON.md`](VRAM_COMPARISON.md) | Measurement boundary and exact results | `vram-comparison.json` |
+| [`EVIDENCE_MANIFEST.json`](EVIDENCE_MANIFEST.json) | Source artifact hashes and generator metadata | Recorded benchmark JSON files |
+
+Regenerate the derived assets from the repository root with:
+
+```bash
+PYTHONPATH=src .venv/bin/python Publishing/Assets/generate_assets.py
+```
+
+Generation requires the `publishing` optional dependency and the `qpdf` command.
+The generator removes volatile PDF metadata and uses a deterministic document
+identifier so the PDF is byte-reproducible.
+
+The historical visual architecture specification remains unchanged in the source repository. It is intentionally not redistributed in the publication packs. The publication architecture diagram is a new project-authored current-architecture asset.
diff --git a/assets/VRAM_COMPARISON.md b/assets/VRAM_COMPARISON.md
new file mode 100644
index 0000000000000000000000000000000000000000..2422502dd474fe83e331f7708c0df79a696a9692
--- /dev/null
+++ b/assets/VRAM_COMPARISON.md
@@ -0,0 +1,46 @@
+# Matched P-cache and retained-KV VRAM comparison
+
+At the largest measured case, canonical P occupied 2.052 MiB while retained KV
+tensors occupied 193.312 MiB. Canonical P was therefore about 94 times smaller
+as a stored representation in this matched test. The two stores are not
+equivalent. Canonical P keeps structured current facts, while retained KV keeps
+recent token-level attention state.
+
+Peak execution memory includes temporary computation. At 1,024 units, the
+incremental peak was 103.491 MiB for P-only and 294.266 MiB for KV-only. These
+peak values should not be confused with the cache tensor sizes above.
+
+The comparison uses one frozen Pythia-1.4B model on one NVIDIA GeForce RTX
+3050 Laptop GPU. Every row uses batch size 1, a float16 base, the same float32
+TTL, greedy generation, eight generated tokens, and the same synthetic prompt
+tokens at each workload size. All mechanisms are warmed once before measurement.
+
+`P-cache only` disables retained runtime KV while leaving normal attention
+working memory and the P-cache TTL path active. `KV only` retains model KV and
+disables P-cache. `P-cache plus KV` enables both. This distinguishes the two
+memory systems without claiming that their contents or purposes are
+interchangeable.
+
+Here, **retained KV** means token-level key-value tensors kept by the runtime.
+**Incremental peak VRAM** means the additional maximum allocated GPU memory
+above the same warmed model baseline.
+
+| Prompt tokens and P slots | Condition | Canonical P | Retained KV | Baseline allocated | Peak allocated | Peak reserved | Incremental peak allocated | Runtime |
+|---:|---|---:|---:|---:|---:|---:|---:|---:|
+| 64 | P-cache only | 0.128 MiB | 0 | 2,717.183 MiB | 2,724.309 MiB | 2,776 MiB | 7.125 MiB | 0.2555 s |
+| 64 | KV only | 0 | 13.312 MiB | 2,717.183 MiB | 2,735.575 MiB | 2,788 MiB | 18.392 MiB | 0.1754 s |
+| 64 | P-cache plus KV | 0.128 MiB | 13.312 MiB | 2,717.183 MiB | 2,735.608 MiB | 2,788 MiB | 18.425 MiB | 0.2045 s |
+| 256 | P-cache only | 0.513 MiB | 0 | 2,717.183 MiB | 2,744.347 MiB | 2,806 MiB | 27.164 MiB | 0.7415 s |
+| 256 | KV only | 0 | 49.312 MiB | 2,717.183 MiB | 2,794.609 MiB | 2,852 MiB | 77.426 MiB | 0.1937 s |
+| 256 | P-cache plus KV | 0.513 MiB | 49.312 MiB | 2,717.183 MiB | 2,794.739 MiB | 2,852 MiB | 77.556 MiB | 0.4156 s |
+| 1,024 | P-cache only | 2.052 MiB | 0 | 2,717.183 MiB | 2,820.674 MiB | 2,938 MiB | 103.491 MiB | 2.7105 s |
+| 1,024 | KV only | 0 | 193.312 MiB | 2,717.183 MiB | 3,011.449 MiB | 3,096 MiB | 294.266 MiB | 0.3753 s |
+| 1,024 | P-cache plus KV | 2.052 MiB | 193.312 MiB | 2,717.183 MiB | 3,011.966 MiB | 3,114 MiB | 294.783 MiB | 1.2812 s |
+
+All nine conditions completed. There were no OOM events, failures, fallbacks,
+or estimated values. CUDA peak allocation includes transient attention, router,
+TTL, output, and allocator behavior. Canonical P bytes and retained KV tensor
+bytes are therefore reported separately from peak deltas.
+
+Raw measurements are in [`vram-comparison.json`](../artifacts/vram-comparison.json). The generated table is
+`vram_comparison.csv`, and the plot is `vram_comparison.svg`.
diff --git a/assets/architecture.mmd b/assets/architecture.mmd
new file mode 100644
index 0000000000000000000000000000000000000000..61ddecceab79dd3feca489dea8d8f4c1bdbac7de
--- /dev/null
+++ b/assets/architecture.mmd
@@ -0,0 +1,24 @@
+flowchart LR
+subgraph Runtime[Model and runtime owned]
+KV[Recent KV]
+LM[Frozen model]
+EXT[History and tool systems]
+end
+subgraph Planner[Planner Cache owned]
+P[P-cache]
+R[Universal router]
+C{Compatibility boundary}
+N[Native P]
+TTL[.ttl semantic/internal]
+LTL[.ltl lexical/output]
+PKG[P-package .ppkg]
+REVIEW[Hidden post-turn review]
+end
+KV --> LM
+LM -. side-channel after visible reply .-> REVIEW --> P
+P --> R --> C
+C --> N --> LM
+C --> TTL --> LM
+C --> LTL --> LM
+PKG -. selected canonical state .-> R
+EXT -. external evidence .-> LM
diff --git a/assets/architecture.pdf b/assets/architecture.pdf
new file mode 100644
index 0000000000000000000000000000000000000000..9fcf3351f4f14b37620dbc05f089d7834a5ba2d5
Binary files /dev/null and b/assets/architecture.pdf differ
diff --git a/assets/architecture.svg b/assets/architecture.svg
new file mode 100644
index 0000000000000000000000000000000000000000..a2e3b38de8b7341761b2c48e1a22bad4761833b6
--- /dev/null
+++ b/assets/architecture.svg
@@ -0,0 +1,48 @@
+
diff --git a/assets/causal_conditions.csv b/assets/causal_conditions.csv
new file mode 100644
index 0000000000000000000000000000000000000000..eef6975c8bc3ca1a2b061d6652137757f2fbb3bf
--- /dev/null
+++ b/assets/causal_conditions.csv
@@ -0,0 +1,17 @@
+condition,selected_state,router_score,router_accepted,gate,alice_logit,bob_logit,generated,kl_from_base,latency_ms
+frozen_base,,,False,0.0,7.02734375,6.12109375, able,-1.646096947638398e-08,17.4486715994135
+historical,user,5.400981426239014,False,0.0,7.02734375,6.12109375, able,-1.646096947638398e-08,19.347908600320807
+invalidated,,,False,0.0,7.02734375,6.12109375, able,-1.646096947638398e-08,17.60459740035003
+p_cache_only,current-task,-0.30040669441223145,False,0.0,7.02734375,6.12109375, able,-1.646096947638398e-08,19.043124400195666
+p_cache_plus_p_package,user,5.590433597564697,True,0.9909488558769226,40.03125,13.859375, Alice,8.359650611877441,19.569637600216083
+p_package_a_relevant,user,5.590433597564697,True,0.9909488558769226,40.03125,13.859375, Alice,8.359650611877441,19.282742799987318
+p_package_b_relevant,user,5.590433597564697,True,0.9605370163917542,14.0,37.15625, Bob,9.265372276306152,18.934362999425502
+p_package_context_creative,user,5.590433597564697,True,0.9605370163917542,14.0,37.15625, Bob,9.265372276306152,19.85094979972928
+p_package_context_technical,user,5.590433597564697,True,0.9909488558769226,40.03125,13.859375, Alice,8.359650611877441,19.666298399533844
+p_package_contradictory_low_confidence,,,False,0.0,7.02734375,6.12109375, able,-1.646096947638398e-08,17.5312320003286
+p_package_irrelevant,,,False,0.0,7.02734375,6.12109375, able,-1.646096947638398e-08,17.46072020032443
+router_disabled,,,False,0.0,7.02734375,6.12109375, able,-1.646096947638398e-08,17.419308600074146
+translator_disabled,user,5.590433597564697,True,0.0,7.02734375,6.12109375, able,-1.646096947638398e-08,18.677392400422832
+translator_oracle_route,user,5.590433597564697,True,0.9909488558769226,40.03125,13.859375, Alice,8.359650611877441,18.915273399761645
+wrong_entity,someone-else,-0.30040669441223145,False,0.0,7.02734375,6.12109375, able,-1.646096947638398e-08,19.29676920044585
+wrong_relation,user,4.078105926513672,False,0.0,7.02734375,6.12109375, able,-1.646096947638398e-08,18.960535999940475
diff --git a/assets/gemma_causal_conditions.csv b/assets/gemma_causal_conditions.csv
new file mode 100644
index 0000000000000000000000000000000000000000..57f8a3739f0a1170db896650265fc184c5a158f0
--- /dev/null
+++ b/assets/gemma_causal_conditions.csv
@@ -0,0 +1,10 @@
+condition,router_accepted,gate,strength,alice_logit,bob_logit,alice_probability,bob_probability,generated,kl_from_base,max_abs_logit_difference_from_base,latency_ms
+correct_alice,True,1.0,1.999999761581421,26.0793858,0.909999311,0.389850411,4.57059089e-12, Alice,0.604321331,11.1606493,320.229877
+correct_bob,True,1.0,-5.000000476837158,7.2261529,26.6650352,2.57553064e-09,0.712961469, Bob,9.63846737,23.8088799,312.696353
+historical,False,0.0,0.0,23.1685543,12.0706482,0.0447660284,6.77936756e-07, the,0,0,404.050531
+invalidated,False,0.0,0.0,23.1685543,12.0706482,0.0447660284,6.77936756e-07, the,0,0,338.130937
+p_disabled,False,0.0,0.0,23.1685543,12.0706482,0.0447660284,6.77936756e-07, the,0,0,354.148533
+router_disabled,False,0.0,0.0,23.1685543,12.0706482,0.0447660284,6.77936756e-07, the,0,0,385.683319
+translator_disabled,True,0.0,0.0,23.1685543,12.0706482,0.0447660284,6.77936756e-07, the,0,0,335.948864
+wrong_entity,False,0.0,0.0,23.1685543,12.0706482,0.0447660284,6.77936756e-07, the,0,0,321.541691
+wrong_relation,False,0.0,0.0,23.1685543,12.0706482,0.0447660284,6.77936756e-07, the,0,0,328.731662
diff --git a/assets/generate_assets.py b/assets/generate_assets.py
new file mode 100644
index 0000000000000000000000000000000000000000..47719a49849ab145461cac69c5132ca887d7eb39
--- /dev/null
+++ b/assets/generate_assets.py
@@ -0,0 +1,363 @@
+"""Generate publication assets from recorded Planner Cache artifacts."""
+
+from __future__ import annotations
+
+import csv
+import hashlib
+import json
+import shutil
+import subprocess
+import tempfile
+from pathlib import Path
+
+import cairosvg
+
+
+ASSETS = Path(__file__).resolve().parent
+PACK_ARTIFACTS = ASSETS.parent / "artifacts"
+ROOT = ASSETS.parent if PACK_ARTIFACTS.is_dir() else ASSETS.parents[1]
+ARTIFACTS = ROOT / "artifacts"
+
+
+def load(name: str):
+ return json.loads((ARTIFACTS / name).read_text())
+
+
+def sha256(path: Path) -> str:
+ digest = hashlib.sha256()
+ with path.open("rb") as handle:
+ for block in iter(lambda: handle.read(1024 * 1024), b""):
+ digest.update(block)
+ return digest.hexdigest()
+
+
+def write_csv(name: str, fields: list[str], rows: list[dict[str, object]]) -> None:
+ with (ASSETS / name).open("w", newline="") as handle:
+ writer = csv.DictWriter(handle, fieldnames=fields)
+ writer.writeheader()
+ writer.writerows(rows)
+
+
+def architecture_svg() -> str:
+ return """
+"""
+
+
+def line_plot_svg(title: str, subtitle: str, series: list[tuple[str, str, list[tuple[float, float]]]], x_label: str, y_label: str) -> str:
+ width = 1100
+ height = 650
+ left = 105
+ right = 55
+ top = 115
+ bottom = 90
+ plot_width = width - left - right
+ plot_height = height - top - bottom
+ all_x = [x for _, _, points in series for x, _ in points]
+ all_y = [y for _, _, points in series for _, y in points]
+ x_min = min(all_x)
+ x_max = max(all_x)
+ y_min = min(0.0, min(all_y))
+ y_max = max(all_y) * 1.08
+
+ def px(value: float) -> float:
+ return left + (value - x_min) / (x_max - x_min) * plot_width
+
+ def py(value: float) -> float:
+ return top + plot_height - (value - y_min) / (y_max - y_min) * plot_height
+
+ parts = [f'')
+ return "\n".join(parts) + "\n"
+
+
+def main() -> None:
+ audit = load("active-system-audit.json")
+ cuda = load("active-system-cuda-attribution.json")
+ split = load("phase-b-split-translator.json")
+ personality = load("phase-b-personality-package.json")
+ gemma = load("gemma4-e4b-q8-causal.json")
+ vram = load("vram-comparison.json")
+
+ router_rows = []
+ for count, values in sorted(audit["router_scaling"]["measurements"].items(), key=lambda item: int(item[0])):
+ router_rows.append({
+ "slots": int(count),
+ "top1_accuracy": values["top1_accuracy"],
+ "top4_recall": values["top4_recall"],
+ "mrr": values["mrr"],
+ "routing_latency_ms": values["latency_seconds"] * 1000,
+ })
+ write_csv("router_scaling.csv", list(router_rows[0]), router_rows)
+
+ ppkg_rows = []
+ for count, values in sorted(audit["ppkg_scaling"].items(), key=lambda item: int(item[0])):
+ ppkg_rows.append({
+ "entries": int(count),
+ "disk_bytes": values["disk_bytes"],
+ "checksum_ms": values["checksum_seconds"] * 1000,
+ "db_open_ms": values["db_open_seconds"] * 1000,
+ "routing_header_ms": values["routing_header_wall_seconds"] * 1000,
+ "row_hydration_ms": values["row_hydration_wall_seconds"] * 1000,
+ "canonical_conversion_ms": values["canonical_conversion_wall_seconds"] * 1000,
+ "candidate_headers": values["candidate_headers"],
+ "entries_loaded": values["entries_loaded"],
+ "logical_bytes_read": values["logical_bytes_read"],
+ "inactive_vram_bytes": values["inactive_vram_bytes"],
+ })
+ write_csv("ppkg_scaling.csv", list(ppkg_rows[0]), ppkg_rows)
+
+ causal_rows = []
+ for condition, values in cuda["causal"].items():
+ causal_rows.append({
+ "condition": condition,
+ "selected_state": values["selected_state"] or "",
+ "router_score": "" if values["router_score"] is None else values["router_score"],
+ "router_accepted": values["router_accepted"],
+ "gate": values["gate"],
+ "alice_logit": values["alice_logit"],
+ "bob_logit": values["bob_logit"],
+ "generated": values["generated"].replace("\n", "\\n"),
+ "kl_from_base": values["kl_from_base"],
+ "latency_ms": cuda["latency_seconds"][condition] * 1000,
+ })
+ write_csv("causal_conditions.csv", list(causal_rows[0]), causal_rows)
+
+ gemma_rows = []
+ for condition, values in gemma["conditions"].items():
+ gemma_rows.append({
+ "condition": condition,
+ "router_accepted": values["router_accepted"],
+ "gate": values["gate"],
+ "strength": values["strength"],
+ "alice_logit": values["alice_logit"],
+ "bob_logit": values["bob_logit"],
+ "alice_probability": values["alice_probability"],
+ "bob_probability": values["bob_probability"],
+ "generated": values["generated"].replace("\n", "\\n"),
+ "kl_from_base": values["kl_from_base"],
+ "max_abs_logit_difference_from_base": values["max_abs_logit_difference_from_base"],
+ "latency_ms": values["latency_ms"],
+ })
+ write_csv("gemma_causal_conditions.csv", list(gemma_rows[0]), gemma_rows)
+
+ vram_rows = []
+ for values in vram["results"]:
+ vram_rows.append({
+ "workload_tokens": values["prompt_tokens"],
+ "condition": values["condition"],
+ "generated_tokens": values["generated_tokens"],
+ "p_cache_slots": values["p_cache_slots"],
+ "p_cache_canonical_bytes": values["p_cache_canonical_bytes"],
+ "retained_kv_cache_bytes": values["retained_kv_cache_bytes"],
+ "baseline_allocated_bytes": values["baseline_allocated_bytes"],
+ "peak_allocated_bytes": values["peak_allocated_bytes"],
+ "peak_reserved_bytes": values["peak_reserved_bytes"],
+ "incremental_peak_allocated_bytes": values["incremental_peak_allocated_bytes"],
+ "incremental_peak_reserved_bytes": values["incremental_peak_reserved_bytes"],
+ "runtime_seconds": values["runtime_seconds"],
+ "failure": "" if values["failure"] is None else values["failure"]["type"],
+ })
+ write_csv("vram_comparison.csv", list(vram_rows[0]), vram_rows)
+
+ architecture = architecture_svg()
+ (ASSETS / "architecture.svg").write_text(architecture)
+ qpdf = shutil.which("qpdf")
+ if qpdf is None:
+ raise RuntimeError("qpdf is required to normalize publication PDF metadata")
+ with tempfile.TemporaryDirectory(prefix="planner-cache-publishing-") as temporary:
+ raw_pdf = Path(temporary) / "architecture.raw.pdf"
+ cairosvg.svg2pdf(bytestring=architecture.encode(), write_to=str(raw_pdf))
+ subprocess.run(
+ [qpdf, "--remove-info", "--remove-metadata", "--deterministic-id", str(raw_pdf), str(ASSETS / "architecture.pdf")],
+ check=True,
+ )
+ (ASSETS / "architecture.mmd").write_text("""flowchart LR
+subgraph Runtime[Model and runtime owned]
+KV[Recent KV]
+LM[Frozen model]
+EXT[History and tool systems]
+end
+subgraph Planner[Planner Cache owned]
+P[P-cache]
+R[Universal router]
+C{Compatibility boundary}
+N[Native P]
+TTL[.ttl semantic/internal]
+LTL[.ltl lexical/output]
+PKG[P-package .ppkg]
+REVIEW[Hidden post-turn review]
+end
+KV --> LM
+LM -. side-channel after visible reply .-> REVIEW --> P
+P --> R --> C
+C --> N --> LM
+C --> TTL --> LM
+C --> LTL --> LM
+PKG -. selected canonical state .-> R
+EXT -. external evidence .-> LM
+""")
+
+ router_points = [(float(row["slots"]), float(row["top1_accuracy"]) * 100) for row in router_rows]
+ legacy = audit["router_scaling"]["immutable_pre_fix_baseline"]
+ legacy_points = [(float(key), float(value) * 100) for key, value in sorted(legacy.items(), key=lambda item: int(item[0]))]
+ (ASSETS / "router_scaling.svg").write_text(line_plot_svg(
+ "Canonical router scaling",
+ "Post-audit identity-safe storage compared with the immutable pre-fix baseline",
+ [("Post-audit top-1", "#2c7a7b", router_points), ("Pre-fix top-1", "#b36a2e", legacy_points)],
+ "Configured P slots",
+ "Top-1 accuracy percent",
+ ))
+
+ ppkg_points = [(float(row["entries"]), float(row["routing_header_ms"])) for row in ppkg_rows]
+ (ASSETS / "ppkg_lookup.svg").write_text(line_plot_svg(
+ "P-package indexed lookup",
+ "Bounded header routing while package contents grow on disk",
+ [("Routing header latency", "#6856a5", ppkg_points)],
+ "Package entries",
+ "Latency ms",
+ ))
+
+ condition_labels = (
+ ("p_cache_only", "P-cache only", "#b36a2e"),
+ ("kv_only", "Retained KV only", "#2c7a7b"),
+ ("p_cache_plus_kv", "P-cache plus KV", "#6856a5"),
+ )
+ vram_series = []
+ for condition, label, color in condition_labels:
+ points = [
+ (
+ float(row["workload_tokens"]),
+ float(row["incremental_peak_allocated_bytes"]) / (1024 * 1024),
+ )
+ for row in vram_rows if row["condition"] == condition
+ ]
+ vram_series.append((label, color, points))
+ (ASSETS / "vram_comparison.svg").write_text(line_plot_svg(
+ "Matched P-cache and retained-KV VRAM",
+ "Frozen Pythia-1.4B, batch 1, float16 base, 8 greedy generated tokens",
+ vram_series,
+ "Prompt tokens and configured P slots",
+ "Incremental peak allocated MiB",
+ ))
+
+ manifest = {
+ "generated_from": {
+ name: sha256(ARTIFACTS / name)
+ for name in (
+ "active-system-audit.json",
+ "active-system-cuda-attribution.json",
+ "phase-b-factorized-representation.json",
+ "phase-b-personality-package.json",
+ "phase-b-split-translator.json",
+ "ppkg-100k-profile.json",
+ "gemma4-e4b-q8-causal.json",
+ "gemma-native-prompt-equivalence.json",
+ "post-turn-memory-review-acceptance.json",
+ "debug-actions-profile.json",
+ "vram-comparison.json",
+ )
+ },
+ "public_binary_artifacts": {
+ name: sha256(ARTIFACTS / name)
+ for name in (
+ "canonical-p-v1.router",
+ "pythia-1.4b-final-layer.ttl",
+ "personality-proof.ppkg",
+ "gemma4-e4b-q8-llama.ltl",
+ )
+ },
+ "selected_configuration": {
+ "base_model": cuda["base_model"],
+ "ttl_parameters": (
+ audit["ttl_profile"]["cpu"]["parameter_count"]
+ if "ttl_profile" in audit
+ else audit["translate_profile"]["cpu"]["parameter_count"]
+ ),
+ "canonical_protocol": "pcm-canonical-p-v1",
+ "router_format": "pcm-canonical-router-v1",
+ "ttl_format": "planner-cache-ttl-v1",
+ "ppkg_format": personality["format"],
+ "ppkg_protocol": personality["protocol"],
+ "selected_attachment": split["selected_variant"],
+ "gemma_model": gemma["model"]["name"],
+ "gemma_ltl_format": "planner-cache-ltl-v1",
+ "gemma_ltl_parameters": 0,
+ },
+ }
+ (ASSETS / "EVIDENCE_MANIFEST.json").write_text(json.dumps(manifest, indent=2, sort_keys=True) + "\n")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/assets/ppkg_lookup.svg b/assets/ppkg_lookup.svg
new file mode 100644
index 0000000000000000000000000000000000000000..3a2d2a2ddd519fecc7bb76b04f7164adddd0cd1d
--- /dev/null
+++ b/assets/ppkg_lookup.svg
@@ -0,0 +1,36 @@
+
diff --git a/assets/ppkg_scaling.csv b/assets/ppkg_scaling.csv
new file mode 100644
index 0000000000000000000000000000000000000000..50a02a3c7d1f32776376cd207006968d14d9fdd1
--- /dev/null
+++ b/assets/ppkg_scaling.csv
@@ -0,0 +1,5 @@
+entries,disk_bytes,checksum_ms,db_open_ms,routing_header_ms,row_hydration_ms,canonical_conversion_ms,candidate_headers,entries_loaded,logical_bytes_read,inactive_vram_bytes
+100,118784,1.0304470015398692,0.20484400010900572,2.207458997872891,0.11649800217128359,0.5500090010173153,4,4,2789,0
+1000,638976,8.523084998159902,0.2957940014312044,14.324849998956779,0.1179409991891589,0.4913600023428444,33,4,6855,0
+10000,5894144,83.29692499683006,0.28353100060485303,57.26086499998928,0.11310199988656677,0.5099939990031999,130,4,20616,0
+100000,58720256,823.8876559989876,0.549067000974901,67.10040300094988,0.11752000136766583,0.4947460001858417,152,4,23715,0
diff --git a/assets/router_scaling.csv b/assets/router_scaling.csv
new file mode 100644
index 0000000000000000000000000000000000000000..4dd2fb71c9ed306da09d40acb1362b07ab8e414e
--- /dev/null
+++ b/assets/router_scaling.csv
@@ -0,0 +1,8 @@
+slots,top1_accuracy,top4_recall,mrr,routing_latency_ms
+4,1.0,1.0,1.0,0.35979299718746915
+20,1.0,1.0,1.0,0.3294770031061489
+64,1.0,1.0,1.0,0.33151000025100075
+128,1.0,1.0,1.0,0.3642720002972055
+256,1.0,1.0,1.0,0.39848600135883316
+512,1.0,1.0,1.0,0.4960180012858473
+1024,1.0,1.0,1.0,0.6750829998054542
diff --git a/assets/router_scaling.svg b/assets/router_scaling.svg
new file mode 100644
index 0000000000000000000000000000000000000000..a45a33f4f63da7a45882c0104a03ac3f8e42bd66
--- /dev/null
+++ b/assets/router_scaling.svg
@@ -0,0 +1,51 @@
+
diff --git a/assets/vram_comparison.csv b/assets/vram_comparison.csv
new file mode 100644
index 0000000000000000000000000000000000000000..5befb91a01592045c2667a99dbef3aa6cebad7e2
--- /dev/null
+++ b/assets/vram_comparison.csv
@@ -0,0 +1,10 @@
+workload_tokens,condition,generated_tokens,p_cache_slots,p_cache_canonical_bytes,retained_kv_cache_bytes,baseline_allocated_bytes,peak_allocated_bytes,peak_reserved_bytes,incremental_peak_allocated_bytes,incremental_peak_reserved_bytes,runtime_seconds,failure
+64,p_cache_only,8,64,134464,0,2849172992,2856644608,2910846976,7471616,4194304,0.2554637499997625,
+64,kv_only,8,0,0,13959168,2849172992,2868457984,2923429888,19284992,16777216,0.17538804200012237,
+64,p_cache_plus_kv,8,64,134464,13959168,2849172992,2868492800,2923429888,19319808,16777216,0.20454967800469603,
+256,p_cache_only,8,256,537856,0,2849172992,2877656576,2942304256,28483584,35651584,0.7414951529935934,
+256,kv_only,8,0,0,51707904,2849172992,2930360320,2990538752,81187328,83886080,0.19367339000746142,
+256,p_cache_plus_kv,8,256,537856,51707904,2849172992,2930496512,2990538752,81323520,83886080,0.41562580699974205,
+1024,p_cache_only,8,1024,2151424,0,2849172992,2957690880,3080716288,108517888,174063616,2.710468404009589,
+1024,kv_only,8,0,0,202702848,2849172992,3157732864,3246391296,308559872,339738624,0.37531103400397114,
+1024,p_cache_plus_kv,8,1024,2151424,202702848,2849172992,3158275072,3265265664,309102080,358612992,1.2812230650015408,
diff --git a/assets/vram_comparison.svg b/assets/vram_comparison.svg
new file mode 100644
index 0000000000000000000000000000000000000000..fb0018507539d62908bb15e37c85a45204a436b4
--- /dev/null
+++ b/assets/vram_comparison.svg
@@ -0,0 +1,45 @@
+
diff --git a/pyproject.toml b/pyproject.toml
new file mode 100644
index 0000000000000000000000000000000000000000..6c974c82cc3bbb3a2167f24d27876db5d99d92ff
--- /dev/null
+++ b/pyproject.toml
@@ -0,0 +1,31 @@
+[build-system]
+requires = ["setuptools>=77"]
+build-backend = "setuptools.build_meta"
+
+[project]
+name = "planner-cache"
+version = "0.1.0"
+description = "Canonical Planner Cache, compatibility layers, and durable personality packages"
+requires-python = ">=3.11"
+dependencies = [
+ "accelerate>=1.12",
+ "numpy>=2.0",
+ "safetensors>=0.5",
+ "torch>=2.7",
+ "transformers>=5.15",
+]
+
+[project.optional-dependencies]
+dev = ["pytest>=9"]
+publishing = ["cairosvg>=2.7"]
+
+[tool.setuptools.packages.find]
+where = ["src"]
+
+[tool.pytest.ini_options]
+testpaths = ["tests"]
+pythonpath = ["src"]
+addopts = "-ra"
+markers = [
+ "slow_cuda: exact CUDA integration regressions using production-scale models",
+]
diff --git a/src/pcm/__init__.py b/src/pcm/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e8ff0191c9235fad2a995c17f58b3d91d12c4f6a
--- /dev/null
+++ b/src/pcm/__init__.py
@@ -0,0 +1,4 @@
+"""Context-efficient adaptive proof model."""
+
+__version__ = "0.1.0"
+
diff --git a/src/pcm/planner/__init__.py b/src/pcm/planner/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..b35e3570518b1fb03622b2c0135852ac1c215863
--- /dev/null
+++ b/src/pcm/planner/__init__.py
@@ -0,0 +1,96 @@
+"""Active Planner Cache architecture exports only."""
+
+from pcm.planner.cache import (
+ CacheFullProtectedError,
+ Freshness,
+ Persistence,
+ PlannerCache,
+ PlannerCacheConfig,
+ SlotSource,
+ SlotType,
+ StateOperation,
+)
+from pcm.planner.canonical import (
+ CANONICAL_P_PROTOCOL,
+ CANONICAL_VALUE_LABELS,
+ CanonicalPConfig,
+ CanonicalPStore,
+ model_config_checksum,
+)
+from pcm.planner.compatibility import (
+ CompatibilityKind,
+ CompatibilityResolution,
+ LTL_EXTENSION,
+ LTL_FORMAT,
+ TTL_EXTENSION,
+ TTL_FORMAT,
+ LexicalTranslationConfig,
+ LexicalTranslationLayer,
+ TensorTranslationLayer,
+ classify_compatibility_artifact,
+ resolve_compatibility,
+ tokenizer_bundle_checksum,
+)
+from pcm.planner.personality import (
+ EvidenceAuthority,
+ EvidenceRecord,
+ FactorizedPersonalityCanonicalizer,
+ PPKG_FORMAT,
+ PPKG_PROTOCOL,
+ PersonalityActivation,
+ PersonalityEntry,
+ PersonalityPackage,
+ PersonalityQuery,
+ PersonalityRouter,
+ PersonalitySelection,
+ PersonalityStatus,
+ PersonalityTranslateSession,
+ PersonalityType,
+ PromotionDecision,
+ PromotionPolicy,
+ evidence_from_p_cache,
+ merge_active_personality_with_p_cache,
+)
+from pcm.planner.pythia_split_translate import (
+ PythiaSplitTranslatedModel,
+ pythia_model_identifier,
+)
+from pcm.planner.interactive_session import (
+ CanonicalQueryIntent,
+ CanonicalStateManager,
+ MutationIntent,
+ PersonalityManager,
+ SessionRecorder,
+)
+from pcm.planner.memory_review import (
+ MEMORY_REVIEW_SCHEMA,
+ REVIEW_CONFIDENCE_FLOOR,
+ REVIEW_FORMAT,
+ PostTurnMemoryReviewer,
+ ReviewedOperation,
+ ValidatedReview,
+ parse_review,
+ review_prompt,
+ validate_review,
+)
+from pcm.planner.interactive_runtimes import (
+ GemmaInteractiveRuntime,
+ GenerationResult,
+ LlamaServerProcess,
+ PythiaInteractiveRuntime,
+ gemma_chat_request_body,
+)
+from pcm.planner.split_translator import (
+ ByteEntityEncoder,
+ CanonicalPRouter,
+ CanonicalValueTranslator,
+ FactorizedCanonicalQuery,
+ FrozenLexicalAnchorProjector,
+ ModelToCanonicalQueryProjector,
+ RouterConfig,
+ SplitInjectionGate,
+ SplitPTranslatePackage,
+ SplitTranslateConfig,
+)
+
+__all__ = [name for name in globals() if not name.startswith("_")]
diff --git a/src/pcm/planner/cache.py b/src/pcm/planner/cache.py
new file mode 100644
index 0000000000000000000000000000000000000000..a8b182697419baf464a6283ff888dd8aa5843096
--- /dev/null
+++ b/src/pcm/planner/cache.py
@@ -0,0 +1,377 @@
+"""Fixed-allocation first-class planner state cache."""
+
+from __future__ import annotations
+
+from dataclasses import dataclass
+from enum import IntEnum
+import math
+from typing import Iterable
+
+import torch
+from torch import Tensor
+import torch.nn.functional as F
+
+
+class StateOperation(IntEnum):
+ KEEP = 0
+ CREATE = 1
+ MODIFY = 2
+ MERGE = 3
+ INVALIDATE = 4
+ IGNORE = 5
+
+
+class SlotType(IntEnum):
+ GOAL = 0
+ ENTITY = 1
+ FACT = 2
+ HYPOTHESIS = 3
+ CONSTRAINT = 4
+ TASK = 5
+ LATENT = 6
+ EXTERNAL = 7
+
+
+class Freshness(IntEnum):
+ FRESH = 0
+ STALE = 1
+ UNKNOWN = 2
+
+
+class Persistence(IntEnum):
+ PERMANENT = 0
+ DURABLE = 1
+ SESSION = 2
+ EXTERNAL = 3
+ VOLATILE = 4
+
+
+class SlotSource(IntEnum):
+ CONVERSATION = 0
+ RETRIEVAL = 1
+ CORRECTION = 2
+ TOOL = 3
+ INFERENCE = 4
+
+
+class CacheFullProtectedError(RuntimeError):
+ """Raised when every physical slot is occupied by permanent state."""
+
+
+@dataclass(frozen=True)
+class PlannerCacheConfig:
+ slots: int = 128
+ width: int = 512
+ dtype: torch.dtype = torch.float16
+ device: str | torch.device = "cpu"
+ merge_similarity: float = 0.92
+
+ def __post_init__(self) -> None:
+ if self.slots <= 0 or self.width <= 0:
+ raise ValueError("planner slots and width must be positive")
+ if not -1.0 <= self.merge_similarity <= 1.0:
+ raise ValueError("merge_similarity must be between -1 and 1")
+
+
+class PlannerCache:
+ """Preallocated planner values and metadata mutated strictly in place."""
+
+ def __init__(self, config: PlannerCacheConfig) -> None:
+ self.config = config
+ device = torch.device(config.device)
+ self.values = torch.zeros((config.slots, config.width), dtype=config.dtype, device=device)
+ self.valid = torch.zeros(config.slots, dtype=torch.bool, device=device)
+ self.slot_type = torch.full((config.slots,), int(SlotType.LATENT), dtype=torch.int8, device=device)
+ self.confidence = torch.zeros(config.slots, dtype=torch.float32, device=device)
+ self.importance = torch.zeros(config.slots, dtype=torch.float32, device=device)
+ self.freshness = torch.full((config.slots,), int(Freshness.UNKNOWN), dtype=torch.int8, device=device)
+ self.persistence = torch.full((config.slots,), int(Persistence.VOLATILE), dtype=torch.int8, device=device)
+ self.last_updated = torch.zeros(config.slots, dtype=torch.int64, device=device)
+ self.source = torch.full((config.slots,), int(SlotSource.INFERENCE), dtype=torch.int8, device=device)
+ self.labels: list[str | None] = [None] * config.slots
+ self._clock = 0
+
+ @property
+ def device(self) -> torch.device:
+ return self.values.device
+
+ def allocation_signature(self) -> tuple[tuple[int, tuple[int, ...]], ...]:
+ """Stable identity/shape signature for physical-allocation tests."""
+ tensors = (
+ self.values,
+ self.valid,
+ self.slot_type,
+ self.confidence,
+ self.importance,
+ self.freshness,
+ self.persistence,
+ self.last_updated,
+ self.source,
+ )
+ return tuple((tensor.data_ptr(), tuple(tensor.shape)) for tensor in tensors)
+
+ @property
+ def occupied(self) -> int:
+ return int(self.valid.sum().item())
+
+ def _tick(self) -> int:
+ self._clock += 1
+ return self._clock
+
+ def _value(self, value: Tensor) -> Tensor:
+ value = value.detach().to(device=self.device, dtype=self.config.dtype)
+ if value.shape != (self.config.width,):
+ raise ValueError(f"planner value must have shape ({self.config.width},)")
+ return value
+
+ def _require_valid(self, index: int) -> None:
+ if not 0 <= index < self.config.slots or not bool(self.valid[index]):
+ raise IndexError(f"planner slot {index} is not valid")
+
+ def _write_metadata(
+ self,
+ index: int,
+ *,
+ slot_type: SlotType,
+ confidence: float,
+ importance: float,
+ freshness: Freshness,
+ persistence: Persistence,
+ source: SlotSource,
+ label: str | None,
+ ) -> None:
+ self._validate_score("confidence", confidence)
+ self._validate_score("importance", importance)
+ slot_type = SlotType(slot_type)
+ freshness = Freshness(freshness)
+ persistence = Persistence(persistence)
+ source = SlotSource(source)
+ if label is not None and not isinstance(label, str):
+ raise TypeError("planner label must be a string or None")
+ self.slot_type[index] = int(slot_type)
+ self.confidence[index] = confidence
+ self.importance[index] = importance
+ self.freshness[index] = int(freshness)
+ self.persistence[index] = int(persistence)
+ self.source[index] = int(source)
+ self.last_updated[index] = self._tick()
+ self.labels[index] = label
+ self.valid[index] = True
+
+ @staticmethod
+ def _validate_score(name: str, value: float) -> None:
+ if not isinstance(value, (int, float)) or not math.isfinite(float(value)):
+ raise ValueError(f"{name} must be a finite number in [0, 1]")
+ if not 0.0 <= float(value) <= 1.0:
+ raise ValueError(f"{name} must be in [0, 1]")
+
+ def _merge_candidate(
+ self, value: Tensor, slot_type: SlotType, merge_mask: Tensor | None = None
+ ) -> int | None:
+ compatible = self.valid & (self.slot_type == int(slot_type))
+ if merge_mask is not None:
+ merge_mask = merge_mask.detach().to(device=self.device, dtype=torch.bool)
+ if merge_mask.shape != self.valid.shape:
+ raise ValueError("merge mask must match the planner slot shape")
+ compatible &= merge_mask
+ indices = compatible.nonzero(as_tuple=False).flatten()
+ if indices.numel() == 0:
+ return None
+ candidates = self.values.index_select(0, indices).float()
+ similarities = F.cosine_similarity(candidates, value.float().unsqueeze(0), dim=-1)
+ best = int(similarities.argmax().item())
+ if float(similarities[best]) < self.config.merge_similarity:
+ return None
+ return int(indices[best].item())
+
+ def _eviction_candidate(self) -> int:
+ candidates = self.valid & (self.persistence != int(Persistence.PERMANENT))
+ indices = candidates.nonzero(as_tuple=False).flatten()
+ if indices.numel() == 0:
+ raise CacheFullProtectedError("all planner slots are permanent")
+ age = (self._clock + 1 - self.last_updated.index_select(0, indices)).float()
+ stale_bonus = (self.freshness.index_select(0, indices) != int(Freshness.FRESH)).float()
+ persistence_cost = torch.tensor(
+ [4.0, 3.0, 2.0, 1.0, 0.0], device=self.device
+ ).index_select(0, self.persistence.index_select(0, indices).long())
+ keep_score = (
+ 4.0 * self.importance.index_select(0, indices)
+ + self.confidence.index_select(0, indices)
+ + persistence_cost
+ - stale_bonus
+ - age * 1e-6
+ )
+ return int(indices[int(keep_score.argmin().item())].item())
+
+ @staticmethod
+ def _admission_score(
+ *,
+ importance: float,
+ confidence: float,
+ freshness: Freshness,
+ persistence: Persistence,
+ ) -> float:
+ persistence_cost = (4.0, 3.0, 2.0, 1.0, 0.0)[int(persistence)]
+ stale_cost = 0.0 if freshness == Freshness.FRESH else 1.0
+ return 4.0 * importance + confidence + persistence_cost - stale_cost
+
+ def _slot_admission_score(self, index: int) -> float:
+ age = (self._clock + 1 - int(self.last_updated[index])) * 1e-6
+ return self._admission_score(
+ importance=float(self.importance[index]),
+ confidence=float(self.confidence[index]),
+ freshness=Freshness(int(self.freshness[index])),
+ persistence=Persistence(int(self.persistence[index])),
+ ) - age
+
+ def create(
+ self,
+ value: Tensor,
+ *,
+ slot_type: SlotType = SlotType.LATENT,
+ confidence: float = 1.0,
+ importance: float = 0.5,
+ freshness: Freshness = Freshness.FRESH,
+ persistence: Persistence = Persistence.SESSION,
+ source: SlotSource = SlotSource.CONVERSATION,
+ label: str | None = None,
+ merge_mask: Tensor | None = None,
+ ) -> tuple[int, StateOperation]:
+ value = self._value(value)
+ self._validate_score("confidence", confidence)
+ self._validate_score("importance", importance)
+ slot_type = SlotType(slot_type)
+ freshness = Freshness(freshness)
+ persistence = Persistence(persistence)
+ source = SlotSource(source)
+ merge_index = self._merge_candidate(value, slot_type, merge_mask)
+ if merge_index is not None:
+ self.merge((merge_index,), value=value, confidence=confidence, source=source)
+ self.importance[merge_index] = max(
+ float(self.importance[merge_index]), importance
+ )
+ self.persistence[merge_index] = min(
+ int(self.persistence[merge_index]), int(persistence)
+ )
+ if label is not None:
+ self.labels[merge_index] = label
+ return merge_index, StateOperation.MERGE
+ free = (~self.valid).nonzero(as_tuple=False).flatten()
+ operation = StateOperation.CREATE
+ if free.numel():
+ index = int(free[0].item())
+ else:
+ index = self._eviction_candidate()
+ incoming_score = self._admission_score(
+ importance=importance,
+ confidence=confidence,
+ freshness=freshness,
+ persistence=persistence,
+ )
+ if incoming_score <= self._slot_admission_score(index):
+ return -1, StateOperation.IGNORE
+ self.invalidate(index)
+ self.values[index].copy_(value)
+ self._write_metadata(
+ index,
+ slot_type=slot_type,
+ confidence=confidence,
+ importance=importance,
+ freshness=freshness,
+ persistence=persistence,
+ source=source,
+ label=label,
+ )
+ return index, operation
+
+ def keep(self, index: int, *, confidence: float | None = None) -> int:
+ self._require_valid(index)
+ if confidence is not None:
+ self._validate_score("confidence", confidence)
+ self.confidence[index] = confidence
+ self.last_updated[index] = self._tick()
+ return index
+
+ def modify(
+ self,
+ index: int,
+ value: Tensor,
+ *,
+ confidence: float | None = None,
+ freshness: Freshness = Freshness.FRESH,
+ source: SlotSource | None = None,
+ ) -> int:
+ self._require_valid(index)
+ freshness = Freshness(freshness)
+ if source is not None:
+ source = SlotSource(source)
+ if confidence is not None:
+ self._validate_score("confidence", confidence)
+ # A model inference is lower-authority than an explicit user
+ # correction and cannot silently overwrite it.
+ if (
+ source == SlotSource.INFERENCE
+ and int(self.source[index]) == int(SlotSource.CORRECTION)
+ ):
+ self.last_updated[index] = self._tick()
+ return index
+ self.values[index].copy_(self._value(value))
+ if confidence is not None:
+ self.confidence[index] = confidence
+ self.freshness[index] = int(freshness)
+ if source is not None:
+ self.source[index] = int(source)
+ self.last_updated[index] = self._tick()
+ return index
+
+ def merge(
+ self,
+ indices: Iterable[int],
+ *,
+ value: Tensor | None = None,
+ confidence: float | None = None,
+ source: SlotSource = SlotSource.INFERENCE,
+ ) -> int:
+ indices = tuple(dict.fromkeys(indices))
+ if not indices:
+ raise ValueError("merge requires at least one slot")
+ for index in indices:
+ self._require_valid(index)
+ target = max(indices, key=lambda index: float(self.importance[index]))
+ merged = self._value(value) if value is not None else self.values[list(indices)].float().mean(0).to(self.config.dtype)
+ self.values[target].copy_(merged)
+ if confidence is None:
+ confidence = max(float(self.confidence[index]) for index in indices)
+ self._validate_score("confidence", confidence)
+ self.confidence[target] = confidence
+ self.importance[target] = max(float(self.importance[index]) for index in indices)
+ self.freshness[target] = int(Freshness.FRESH)
+ self.source[target] = int(source)
+ self.last_updated[target] = self._tick()
+ for index in indices:
+ if index != target:
+ self.invalidate(index)
+ return target
+
+ def invalidate(self, index: int) -> int:
+ self._require_valid(index)
+ self.valid[index] = False
+ self.values[index].zero_()
+ self.labels[index] = None
+ self.last_updated[index] = self._tick()
+ return index
+
+ def apply(self, operation: StateOperation, **kwargs):
+ if operation == StateOperation.KEEP:
+ return self.keep(**kwargs)
+ if operation == StateOperation.CREATE:
+ return self.create(**kwargs)
+ if operation == StateOperation.MODIFY:
+ return self.modify(**kwargs)
+ if operation == StateOperation.MERGE:
+ return self.merge(**kwargs)
+ if operation == StateOperation.INVALIDATE:
+ return self.invalidate(**kwargs)
+ if operation == StateOperation.IGNORE:
+ return None
+ raise ValueError(f"unsupported planner operation: {operation}")
diff --git a/src/pcm/planner/canonical.py b/src/pcm/planner/canonical.py
new file mode 100644
index 0000000000000000000000000000000000000000..f27667887cc1c21cdcc1bfd0b347a7dce783c5f4
--- /dev/null
+++ b/src/pcm/planner/canonical.py
@@ -0,0 +1,220 @@
+"""Model-independent canonical Planner Cache protocol and storage."""
+
+from __future__ import annotations
+
+from dataclasses import dataclass
+import hashlib
+import json
+from pathlib import Path
+
+import torch
+from torch import Tensor
+from safetensors import safe_open
+from safetensors.torch import load_file, save_file
+
+from pcm.planner.cache import PlannerCache, PlannerCacheConfig, SlotType
+
+
+CANONICAL_VALUE_LABELS = tuple(
+ "Alice Bob Clara David Elena Frank Grace Henry Irene James Karen Louis Maria Nancy "
+ "Oscar Peter Queen Robert Sarah Thomas Victor Wendy Xavier London Paris Berlin Rome "
+ "Cairo Tokyo Sydney garden kitchen cellar library forest castle".split()
+)
+
+
+def tensor_state_checksum(state: dict[str, Tensor]) -> str:
+ """Deterministic checksum for portable tensor-only state."""
+ digest = hashlib.sha256()
+ for name, tensor in sorted(state.items()):
+ value = tensor.detach().cpu().contiguous()
+ digest.update(name.encode())
+ digest.update(str(value.dtype).encode())
+ digest.update(json.dumps(list(value.shape)).encode())
+ digest.update(value.reshape(-1).view(torch.uint8).numpy().tobytes())
+ return digest.hexdigest()
+
+
+def canonical_snapshot_checksum(
+ tensors: dict[str, Tensor], metadata: dict[str, str]
+) -> str:
+ digest = hashlib.sha256(tensor_state_checksum(tensors).encode())
+ digest.update(json.dumps(metadata, sort_keys=True, separators=(",", ":")).encode())
+ return digest.hexdigest()
+
+
+@dataclass(frozen=True)
+class CanonicalPConfig:
+ slots: int = 128
+ width: int = 512
+ dtype: torch.dtype = torch.float16
+ device: str | torch.device = "cpu"
+ merge_similarity: float = 0.92
+
+
+class CanonicalPStore:
+ """Fixed P slots whose serialized state has no model-hidden representation."""
+
+ FORMAT = "pcm-canonical-p-v1"
+
+ def __init__(self, config: CanonicalPConfig) -> None:
+ self.config = config
+ self.cache = PlannerCache(PlannerCacheConfig(
+ slots=config.slots,
+ width=config.width,
+ dtype=config.dtype,
+ device=config.device,
+ merge_similarity=config.merge_similarity,
+ ))
+ device = self.cache.device
+ self.entity_id = torch.full((config.slots,), -1, dtype=torch.int64, device=device)
+ self.relation_id = torch.full((config.slots,), -1, dtype=torch.int64, device=device)
+ self.value_id = torch.full((config.slots,), -1, dtype=torch.int64, device=device)
+ self.canonical_metadata_id = torch.full(
+ (config.slots,), -1, dtype=torch.int64, device=device
+ )
+
+ @property
+ def canonical_values(self) -> Tensor:
+ return self.cache.values
+
+ @property
+ def valid(self) -> Tensor:
+ return self.cache.valid
+
+ def create(
+ self,
+ canonical_value: Tensor,
+ *,
+ entity_id: int,
+ relation_id: int,
+ value_id: int,
+ metadata_id: int,
+ slot_type: SlotType = SlotType.FACT,
+ **cache_metadata,
+ ) -> tuple[int, object]:
+ merge_mask = (
+ (self.entity_id == entity_id)
+ & (self.relation_id == relation_id)
+ & self.cache.valid
+ )
+ slot, operation = self.cache.create(
+ canonical_value, slot_type=slot_type, merge_mask=merge_mask, **cache_metadata
+ )
+ if slot >= 0:
+ self.entity_id[slot] = entity_id
+ self.relation_id[slot] = relation_id
+ self.value_id[slot] = value_id
+ self.canonical_metadata_id[slot] = metadata_id
+ return slot, operation
+
+ def allocation_signature(self):
+ fields = (self.entity_id, self.relation_id, self.value_id, self.canonical_metadata_id)
+ return self.cache.allocation_signature() + tuple(
+ (field.data_ptr(), tuple(field.shape)) for field in fields
+ )
+
+ def modify(
+ self,
+ slot: int,
+ canonical_value: Tensor,
+ *,
+ entity_id: int,
+ relation_id: int,
+ value_id: int,
+ metadata_id: int,
+ **cache_metadata,
+ ) -> int:
+ result = self.cache.modify(slot, canonical_value, **cache_metadata)
+ self.entity_id[slot] = entity_id
+ self.relation_id[slot] = relation_id
+ self.value_id[slot] = value_id
+ self.canonical_metadata_id[slot] = metadata_id
+ return result
+
+ def invalidate(self, slot: int) -> int:
+ result = self.cache.invalidate(slot)
+ self.entity_id[slot] = -1
+ self.relation_id[slot] = -1
+ self.value_id[slot] = -1
+ self.canonical_metadata_id[slot] = -1
+ return result
+
+ def save(self, path: str | Path) -> None:
+ tensors = {
+ "canonical_values": self.cache.values.detach().cpu(),
+ "valid": self.cache.valid.detach().cpu(),
+ "slot_type": self.cache.slot_type.detach().cpu(),
+ "confidence": self.cache.confidence.detach().cpu(),
+ "importance": self.cache.importance.detach().cpu(),
+ "freshness": self.cache.freshness.detach().cpu(),
+ "persistence": self.cache.persistence.detach().cpu(),
+ "last_updated": self.cache.last_updated.detach().cpu(),
+ "source": self.cache.source.detach().cpu(),
+ "entity_id": self.entity_id.detach().cpu(),
+ "relation_id": self.relation_id.detach().cpu(),
+ "value_id": self.value_id.detach().cpu(),
+ "canonical_metadata_id": self.canonical_metadata_id.detach().cpu(),
+ }
+ metadata = {
+ "format": self.FORMAT,
+ "config": json.dumps({
+ "slots": self.config.slots,
+ "width": self.config.width,
+ "merge_similarity": self.config.merge_similarity,
+ }),
+ "labels": json.dumps(self.cache.labels),
+ }
+ metadata["content_sha256"] = canonical_snapshot_checksum(tensors, metadata)
+ save_file(tensors, str(Path(path)), metadata=metadata)
+
+ @classmethod
+ def load(
+ cls,
+ path: str | Path,
+ *,
+ device: str | torch.device = "cpu",
+ dtype: torch.dtype = torch.float16,
+ ) -> "CanonicalPStore":
+ path = Path(path)
+ with safe_open(str(path), framework="pt", device="cpu") as handle:
+ metadata = handle.metadata()
+ if metadata.get("format") != cls.FORMAT:
+ raise ValueError("unsupported canonical P snapshot")
+ config = json.loads(metadata["config"])
+ tensors = load_file(str(path), device=str(device))
+ checksum_metadata = {
+ key: value for key, value in metadata.items() if key != "content_sha256"
+ }
+ if canonical_snapshot_checksum(tensors, checksum_metadata) != metadata.get(
+ "content_sha256"
+ ):
+ raise ValueError("canonical P snapshot checksum does not match")
+ result = cls(CanonicalPConfig(
+ slots=config["slots"], width=config["width"], dtype=dtype, device=device,
+ merge_similarity=config.get("merge_similarity", 0.92),
+ ))
+ result.cache.values.copy_(tensors["canonical_values"].to(dtype=dtype))
+ for name in (
+ "valid", "slot_type", "confidence", "importance", "freshness",
+ "persistence", "last_updated", "source",
+ ):
+ getattr(result.cache, name).copy_(tensors[name])
+ for name in ("entity_id", "relation_id", "value_id", "canonical_metadata_id"):
+ getattr(result, name).copy_(tensors[name])
+ result.cache.labels = json.loads(metadata["labels"])
+ result.cache._clock = int(result.cache.last_updated.max())
+ return result
+
+
+CANONICAL_P_PROTOCOL = CanonicalPStore.FORMAT
+
+
+def model_config_checksum(config: object) -> str:
+ payload = config.to_dict() if hasattr(config, "to_dict") else config
+ if isinstance(payload, dict):
+ payload = dict(payload)
+ configured_path = payload.get("_name_or_path")
+ if configured_path and Path(str(configured_path)).is_absolute():
+ payload["_name_or_path"] = Path(str(configured_path)).name
+ encoded = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode("utf-8")
+ return hashlib.sha256(encoded).hexdigest()
diff --git a/src/pcm/planner/chat_cli.py b/src/pcm/planner/chat_cli.py
new file mode 100644
index 0000000000000000000000000000000000000000..553f6ca84f31310831b4315c4a42b48524b5aab1
--- /dev/null
+++ b/src/pcm/planner/chat_cli.py
@@ -0,0 +1,561 @@
+"""Shared Planner Cache chat session and terminal entry point."""
+
+from __future__ import annotations
+
+import argparse
+from dataclasses import asdict
+import json
+from pathlib import Path
+import sys
+import traceback
+
+import torch
+
+from pcm.planner.canonical import CANONICAL_P_PROTOCOL, CanonicalPStore
+from pcm.planner.compatibility import CompatibilityKind, resolve_compatibility
+from pcm.planner.interactive_runtimes import (
+ GenerationResult,
+ GemmaInteractiveRuntime,
+ PythiaInteractiveRuntime,
+ canonical_route,
+)
+from pcm.planner.interactive_session import (
+ CanonicalStateManager,
+ PersonalityManager,
+ SessionRecorder,
+ git_commit,
+ utc_now,
+)
+from pcm.planner.memory_review import PostTurnMemoryReviewer
+from pcm.planner.representation import train_and_probe_representation
+
+
+HELP = """Chat normally by typing any message.
+
+Commands:
+ /help show this help
+ /state show active canonical P-cache entries
+ /personality show promoted personality entries and the last retrieval
+ /events show the most recent Planner Cache events
+ /save checkpoint P-cache and P-package state
+ /quit save and exit
+
+Memory extraction is automatic. These explicit forms are also recognized:
+ The silver key belongs to Alice
+ Alice owns the silver key
+ The silver key is currently in Paris
+ The current status of the silver key is garden
+ remember: silver key.owner=Alice
+ invalidate: silver key.owner
+
+Pythia consumes accepted state through its TTL. Gemma uses an LTL for accepted
+lexical values. Rejected routes remain inert and conversation continues normally.
+"""
+
+
+def parser() -> argparse.ArgumentParser:
+ root = Path(__file__).resolve().parents[3]
+ result = argparse.ArgumentParser(description="Planner Cache interactive terminal")
+ result.add_argument("runtime", choices=("pythia", "gemma"))
+ result.add_argument("--repo-root", type=Path, default=root)
+ result.add_argument("--model", type=Path, required=True)
+ result.add_argument("--adapter", type=Path, required=True)
+ result.add_argument("--router", type=Path, required=True)
+ result.add_argument("--llama-cpp-dir", type=Path)
+ result.add_argument("--tokenizer-bundle", type=Path)
+ result.add_argument("--ppkg", type=Path)
+ result.add_argument("--session-root", type=Path, required=True)
+ result.add_argument("--p-cache", type=Path)
+ result.add_argument("--slots", type=int, default=128)
+ result.add_argument("--context-tokens", type=int)
+ result.add_argument("--max-new-tokens", type=int, default=96)
+ result.add_argument("--temperature", type=float, default=0.7)
+ result.add_argument("--top-p", type=float, default=0.9)
+ result.add_argument("--seed", type=int, default=1234)
+ result.add_argument("--gpu-layers", type=int, default=12)
+ result.add_argument("--threads", type=int, default=8)
+ result.add_argument("--llama-pid-file", type=Path)
+ result.add_argument("--review-model", type=Path)
+ result.add_argument("--review-llama-cpp-dir", type=Path)
+ result.add_argument("--review-gpu-layers", type=int, default=0)
+ result.add_argument("--review-pid-file", type=Path)
+ result.add_argument("--no-logging", action="store_true")
+ return result
+
+
+def require_file(path: Path, label: str) -> Path:
+ if not path.is_file():
+ raise FileNotFoundError(f"{label} not found: {path}")
+ return path
+
+
+def validate_args(args: argparse.Namespace) -> None:
+ if args.runtime == "pythia":
+ if not args.model.is_dir():
+ raise FileNotFoundError(f"pythia model not found: {args.model}")
+ review_model = getattr(args, "review_model", None)
+ review_llama_cpp_dir = getattr(args, "review_llama_cpp_dir", None)
+ if review_model is not None:
+ require_file(review_model, "structured review model")
+ if review_llama_cpp_dir is None:
+ raise ValueError(
+ "--review-llama-cpp-dir is required with --review-model"
+ )
+ require_file(
+ review_llama_cpp_dir / "build/bin/llama-server",
+ "review llama-server",
+ )
+ else:
+ require_file(args.model, "gemma model")
+ expected = ".ttl" if args.runtime == "pythia" else ".ltl"
+ require_file(args.adapter, f"{expected} compatibility artifact")
+ if args.adapter.suffix != expected:
+ raise ValueError(f"{args.runtime} requires a {expected} compatibility artifact")
+ resolution = resolve_compatibility(args.adapter)
+ expected_kind = (
+ CompatibilityKind.TTL if args.runtime == "pythia" else CompatibilityKind.LTL
+ )
+ if resolution.kind is not expected_kind:
+ raise ValueError(
+ f"{args.runtime} cannot use {resolution.kind.value} compatibility"
+ )
+ require_file(args.router, ".router artifact")
+ if args.p_cache is not None:
+ require_file(args.p_cache, "P-cache snapshot")
+ if args.runtime == "gemma":
+ if args.llama_cpp_dir is None:
+ raise ValueError("--llama-cpp-dir is required for Gemma")
+ if args.tokenizer_bundle is None or not args.tokenizer_bundle.is_dir():
+ raise FileNotFoundError("--tokenizer-bundle is required for Gemma")
+ require_file(args.llama_cpp_dir / "build/bin/llama-server", "llama-server")
+ require_file(args.llama_cpp_dir / "build/bin/llama-cli", "llama-cli")
+
+
+def build_runtime(args: argparse.Namespace):
+ common = {
+ "model_path": args.model,
+ "adapter_path": args.adapter,
+ "router_path": args.router,
+ "max_new_tokens": args.max_new_tokens,
+ "temperature": args.temperature,
+ "top_p": args.top_p,
+ "seed": args.seed,
+ }
+ if args.runtime == "pythia":
+ return PythiaInteractiveRuntime(
+ **common,
+ max_context_tokens=args.context_tokens or 1024,
+ review_model_path=getattr(args, "review_model", None),
+ review_llama_cpp_dir=getattr(args, "review_llama_cpp_dir", None),
+ review_gpu_layers=getattr(args, "review_gpu_layers", 0),
+ review_pid_file=getattr(args, "review_pid_file", None),
+ )
+ return GemmaInteractiveRuntime(
+ **common,
+ llama_cpp_dir=args.llama_cpp_dir,
+ tokenizer_bundle=args.tokenizer_bundle,
+ max_context_tokens=args.context_tokens or 4096,
+ gpu_layers=args.gpu_layers,
+ threads=args.threads,
+ pid_file=args.llama_pid_file,
+ )
+
+
+def display_json(value: object) -> None:
+ print(json.dumps(value, indent=2, sort_keys=True, ensure_ascii=False))
+
+
+def save_session_state(
+ recorder: SessionRecorder,
+ state: CanonicalStateManager,
+ personality: PersonalityManager | None,
+ *,
+ reason: str,
+) -> None:
+ snapshot = recorder.directory / "p-cache.safetensors"
+ if recorder.enabled:
+ state.store.save(snapshot)
+ state.save_runtime_metadata(recorder.directory / "p-cache-runtime.json")
+ checksum = personality.checkpoint() if personality is not None else None
+ recorder.save_event(
+ reason,
+ p_cache_snapshot=str(snapshot) if recorder.enabled else None,
+ ppkg_checksum=checksum,
+ )
+
+
+def final_payload(
+ state: CanonicalStateManager,
+ personality: PersonalityManager | None,
+) -> dict[str, object]:
+ personality_page = (
+ {
+ "entries": [], "total_active": 0, "returned": 0,
+ "limit": 100, "offset": 0, "truncated": False,
+ }
+ if personality is None
+ else personality.visible_entry_page(limit=100)
+ )
+ return {
+ "canonical_p_protocol": CANONICAL_P_PROTOCOL,
+ "active_p_cache_entries": state.snapshot(),
+ "ppkg_path": None if personality is None else str(personality.path),
+ "ppkg_mutations": [] if personality is None else personality.mutations,
+ "promoted_personality_entries": personality_page.pop("entries"),
+ "promoted_personality_page": personality_page,
+ }
+
+
+class PlannerChatSession:
+ """One active Planner Cache session shared by terminal and web front ends."""
+
+ def __init__(self, args: argparse.Namespace) -> None:
+ validate_args(args)
+ self.args = args
+ representation, _config, probe = train_and_probe_representation()
+ store = (
+ CanonicalPStore.load(args.p_cache, dtype=torch.float32)
+ if args.p_cache is not None else None
+ )
+ self.state = CanonicalStateManager(representation, slots=args.slots, store=store)
+ if args.p_cache is not None:
+ self.state.load_runtime_metadata(
+ args.p_cache.with_name("p-cache-runtime.json")
+ )
+ self.personality = (
+ PersonalityManager(args.ppkg, representation)
+ if args.ppkg is not None else None
+ )
+ try:
+ self.runtime = build_runtime(args)
+ except Exception:
+ if self.personality is not None:
+ self.personality.close()
+ raise
+ self.reviewer = PostTurnMemoryReviewer(self.runtime.review_memory)
+ metadata = {
+ **self.runtime.metadata(),
+ "canonical_p_protocol_version": CANONICAL_P_PROTOCOL,
+ "ppkg_path": (
+ None if self.personality is None
+ else str(self.personality.path.resolve())
+ ),
+ "git_commit": git_commit(args.repo_root),
+ "launch_arguments": vars(args) | {
+ key: None if value is None else str(value)
+ for key, value in vars(args).items() if isinstance(value, Path)
+ },
+ "canonical_representation": {
+ "recipe": "train_and_probe_representation",
+ "seed": 97,
+ "training_steps": 600,
+ "held_out_p_only_state_recovery": probe["p_only_state_recovery"],
+ },
+ }
+ self.recorder = SessionRecorder(
+ args.session_root, args.runtime, metadata,
+ enabled=not args.no_logging,
+ )
+ self.history: list[tuple[str, str]] = []
+ self._pending_review: tuple[str, str, list[tuple[str, str]]] | None = None
+ self.closed = False
+
+ @property
+ def title(self) -> str:
+ return (
+ "Planner Cache — Pythia-1.4B"
+ if self.args.runtime == "pythia"
+ else "Planner Cache — Gemma4 E4B"
+ )
+
+ def command(
+ self,
+ command: str,
+ *,
+ source: str = "terminal",
+ personality_limit: int = 100,
+ personality_offset: int = 0,
+ ) -> object:
+ command = command.strip().casefold()
+ self.recorder.event("SESSION_COMMAND", source=source, command=command)
+ if command == "/help":
+ return HELP
+ if command == "/state":
+ return self.state.snapshot()
+ if command == "/personality":
+ page = (
+ {
+ "entries": [], "total_active": 0, "returned": 0,
+ "limit": personality_limit, "offset": personality_offset,
+ "truncated": False,
+ }
+ if self.personality is None
+ else self.personality.visible_entry_page(
+ limit=personality_limit, offset=personality_offset,
+ )
+ )
+ return {
+ "promoted": page.pop("entries"),
+ "page": page,
+ "last_retrieval": (
+ None
+ if self.personality is None
+ or self.personality.last_selection is None
+ else {
+ "route": asdict(self.personality.last_selection.route),
+ "entries": [
+ asdict(entry)
+ for entry in self.personality.last_selection.entries
+ ],
+ }
+ ),
+ }
+ if command == "/events":
+ return list(self.recorder.recent_events)
+ if command == "/save":
+ save_session_state(
+ self.recorder, self.state, self.personality, reason="explicit"
+ )
+ return "Session state saved."
+ if command == "/quit":
+ return "quit"
+ return "Unknown command. Type /help."
+
+ def chat(
+ self,
+ message: str,
+ *,
+ raw_messages: list[dict[str, object]] | None = None,
+ ) -> GenerationResult:
+ if self._pending_review is not None:
+ raise RuntimeError("previous post-turn memory review is incomplete")
+ self.recorder.turn += 1
+ self.recorder.transcript(
+ role="user", text=message, model=self.runtime.model_id,
+ runtime=self.runtime.runtime,
+ )
+ self.recorder.event(
+ "P_STATE_BEFORE", source="p_cache", entries=self.state.snapshot()
+ )
+ mutations = self.state.extract_manual_mutations(message)
+ if mutations:
+ self.state.apply(mutations, self.recorder)
+ else:
+ self.recorder.event(
+ "P_IGNORE", source="p_cache",
+ reason="no deterministic manual override before generation",
+ )
+ if self.personality is not None:
+ evidence = self.personality.extract_evidence(
+ message, turn=self.recorder.turn, timestamp=utc_now(),
+ )
+ if evidence is not None:
+ self.personality.ingest(evidence, self.recorder)
+ personality_store = self.personality.query(message, self.recorder)
+ else:
+ personality_store = None
+ query = self.state.infer_query(message)
+ if (
+ query.entity is None
+ and personality_store is not None
+ and personality_store.cache.occupied
+ ):
+ query = type(query)(
+ entity="user", relation_id=0, relation="response_style",
+ reason="accepted context-specific P-package state",
+ )
+ open_values = bool(getattr(self.runtime, "supports_open_values", False))
+ translation_store = self.state.translation_store(
+ include_open_values=open_values
+ )
+ if query.entity is not None and query.relation_id is not None:
+ full_route, full_candidates = canonical_route(
+ self.runtime.router, self.runtime.encoder, self.state.store, query,
+ )
+ if full_route is not None and full_route.has_valid:
+ selected_slot = int(full_route.indices[0, 0])
+ accepted = bool(full_route.accepted[0])
+ if (
+ accepted
+ and not open_values
+ and not self.state.translator_compatible.get(selected_slot, True)
+ ):
+ self.recorder.event(
+ "ROUTER_QUERY", source="canonical_compatibility_precheck",
+ entity=query.entity, relation=query.relation,
+ )
+ self.recorder.event(
+ "ROUTER_CANDIDATES", source="p_cache",
+ candidates=full_candidates,
+ )
+ self.recorder.event(
+ "ROUTER_ACCEPT", source="p_cache",
+ score=float(full_route.scores[0, 0].detach()),
+ selected_state=self.state.entry(selected_slot),
+ )
+ self.recorder.event(
+ "TTL_DISABLE", source="ttl",
+ reason=(
+ "canonical value is outside this TTL's supported vocabulary"
+ ),
+ selected_state=self.state.entry(selected_slot),
+ )
+ self.recorder.event(
+ "MODEL_GENERATION_START", source="model", model=self.runtime.model_id,
+ runtime=self.runtime.runtime, query=asdict(query),
+ )
+ started = utc_now()
+ try:
+ result = self.runtime.generate(
+ message, self.history, translation_store, personality_store, query,
+ emit=self.recorder.event if self.recorder.enabled else None,
+ raw_messages=raw_messages,
+ )
+ except Exception as error:
+ self.recorder.event(
+ "ERROR", source="model", error_type=type(error).__name__,
+ message=str(error), traceback=traceback.format_exc(),
+ )
+ raise
+ self.recorder.event(
+ "MODEL_GENERATION_END", source="model", started_at=started,
+ latency_seconds=result.latency_seconds,
+ input_tokens=result.input_tokens, output_tokens=result.output_tokens,
+ diagnostics=result.diagnostics,
+ )
+ self.recorder.transcript(
+ role="assistant", text=result.text, model=self.runtime.model_id,
+ runtime=self.runtime.runtime, latency_seconds=result.latency_seconds,
+ input_tokens=result.input_tokens, output_tokens=result.output_tokens,
+ )
+ model_candidates = self.state.extract_mutations(result.text)
+ for candidate in model_candidates:
+ self.recorder.event(
+ "P_IGNORE", source="model_output", candidate=asdict(candidate),
+ reason="model-generated state is not authoritative without user evidence",
+ )
+ self._pending_review = (message, result.text, list(self.history))
+ return result
+
+ def complete_turn_review(self) -> None:
+ """Finish the hidden review before another visible turn may start."""
+ if self._pending_review is None:
+ return
+ message, response, prior_history = self._pending_review
+ self._pending_review = None
+ self.reviewer.run(
+ user_message=message,
+ assistant_response=response,
+ recent_context=prior_history,
+ state=self.state,
+ recorder=self.recorder,
+ )
+ self.history.append((message, response))
+ self.recorder.event(
+ "P_STATE_AFTER", source="p_cache", entries=self.state.snapshot()
+ )
+
+ def close(self, *, reason: str = "normal") -> None:
+ if self.closed:
+ return
+ self.closed = True
+ if self._pending_review is not None:
+ self.complete_turn_review()
+ try:
+ self.runtime.close()
+ except Exception as error:
+ self.recorder.event(
+ "ERROR", source="runtime_cleanup",
+ error_type=type(error).__name__, message=str(error),
+ )
+ try:
+ save_session_state(
+ self.recorder, self.state, self.personality, reason="final"
+ )
+ except Exception as error:
+ self.recorder.event(
+ "ERROR", source="session_save",
+ error_type=type(error).__name__, message=str(error),
+ )
+ try:
+ payload = final_payload(self.state, self.personality)
+ except Exception as error:
+ self.recorder.event(
+ "ERROR", source="final_state",
+ error_type=type(error).__name__, message=str(error),
+ )
+ payload = {
+ "canonical_p_protocol": CANONICAL_P_PROTOCOL,
+ "active_p_cache_entries": self.state.snapshot(),
+ "ppkg_mutations": [],
+ "final_state_error": str(error),
+ }
+ self.recorder.finalize(payload, reason=reason)
+ if self.personality is not None:
+ try:
+ self.personality.close()
+ except Exception:
+ pass
+
+
+def run(args: argparse.Namespace) -> int:
+ session = None
+ reason = "normal"
+ try:
+ session = PlannerChatSession(args)
+ title = (
+ "Planner Cache — Pythia-1.4B"
+ if args.runtime == "pythia"
+ else "Planner Cache — Gemma4 E4B"
+ )
+ print(f"\n{title}\n")
+ while True:
+ try:
+ message = input("You: ")
+ except EOFError:
+ reason = "eof"
+ break
+ stripped = message.strip()
+ if not stripped:
+ continue
+ if stripped.startswith("/"):
+ command = stripped.casefold()
+ if command == "/quit":
+ session.command(command)
+ reason = "quit"
+ break
+ value = session.command(command)
+ if isinstance(value, str):
+ print(value)
+ else:
+ display_json(value)
+ continue
+ try:
+ result = session.chat(message)
+ except KeyboardInterrupt:
+ reason = "ctrl-c"
+ raise
+ except Exception as error:
+ print(f"Generation failed: {error}", file=sys.stderr)
+ continue
+ print(f"Assistant: {result.text}")
+ session.complete_turn_review()
+ except KeyboardInterrupt:
+ reason = "ctrl-c"
+ print("\nStopping.")
+ finally:
+ if session is not None:
+ session.close(reason=reason)
+ return 0
+
+
+def main() -> None:
+ try:
+ raise SystemExit(run(parser().parse_args()))
+ except (FileNotFoundError, ValueError, RuntimeError) as error:
+ print(f"Planner Cache startup failed: {error}", file=sys.stderr)
+ raise SystemExit(2) from error
+
+
+if __name__ == "__main__":
+ main()
diff --git a/src/pcm/planner/compatibility.py b/src/pcm/planner/compatibility.py
new file mode 100644
index 0000000000000000000000000000000000000000..d962f23080582edac9d902ae5b158a926dc67db8
--- /dev/null
+++ b/src/pcm/planner/compatibility.py
@@ -0,0 +1,317 @@
+"""Explicit model compatibility boundaries for canonical Planner Cache state.
+
+TTL exposes canonical state inside a model forward path. LTL is intentionally
+weaker and controls lexical output at a tokenizer or runtime boundary.
+"""
+
+from __future__ import annotations
+
+from dataclasses import asdict, dataclass
+from enum import Enum
+import hashlib
+import json
+from pathlib import Path
+import warnings
+
+import torch
+from safetensors import safe_open
+from safetensors._safetensors_rust import SafetensorError
+from safetensors.torch import load_file, save_file
+
+from pcm.planner.canonical import CANONICAL_P_PROTOCOL
+from pcm.planner.split_translator import (
+ SPLIT_TRANSLATE_FORMAT,
+ SplitPTranslatePackage,
+ SplitTranslateConfig,
+ tensor_checksum,
+)
+
+
+TTL_FORMAT = "planner-cache-ttl-v1"
+LTL_FORMAT = "planner-cache-ltl-v1"
+TTL_EXTENSION = ".ttl"
+LTL_EXTENSION = ".ltl"
+FORBIDDEN_ADAPTER_FIELDS = (
+ "base_model",
+ "conversation",
+ "p_cache",
+ "canonical_values",
+ "optimizer",
+)
+
+
+class CompatibilityKind(str, Enum):
+ NATIVE = "native"
+ TTL = "ttl"
+ LTL = "ltl"
+
+
+@dataclass(frozen=True)
+class CompatibilityResolution:
+ kind: CompatibilityKind
+ support_level: str
+ artifact: Path | None
+
+
+class TensorTranslationLayer(SplitPTranslatePackage):
+ """Semantic/internal compatibility module backed by the Pythia TTL.
+
+ The learned module is unchanged from the proven split translator. The new
+ container identifies its stronger semantic contract and rejects LTL files.
+ """
+
+ adapter_class = CompatibilityKind.TTL.value
+ support_level = "semantic/internal"
+ extension = TTL_EXTENSION
+
+ def save(self, path: str | Path) -> None:
+ path = Path(path)
+ if path.suffix != TTL_EXTENSION:
+ raise ValueError(f"TTL artifacts must use the {TTL_EXTENSION} extension")
+ state = {name: value.detach().cpu() for name, value in self.state_dict().items()}
+ if any(field in name.casefold() for name in state for field in FORBIDDEN_ADAPTER_FIELDS):
+ raise ValueError("TTL contains forbidden model or conversation state")
+ config = asdict(self.config)
+ config.pop("format", None)
+ manifest = {
+ "format": TTL_FORMAT,
+ "adapter_class": self.adapter_class,
+ "support_level": self.support_level,
+ "canonical_protocol": self.config.canonical_protocol,
+ "config": config,
+ "weights_sha256": tensor_checksum(state),
+ }
+ save_file(state, str(path), metadata={
+ "manifest": json.dumps(manifest, sort_keys=True, separators=(",", ":")),
+ })
+
+ @classmethod
+ def load(
+ cls,
+ path: str | Path,
+ *,
+ device: str | torch.device = "cpu",
+ dtype: torch.dtype = torch.float32,
+ allow_legacy: bool = True,
+ ) -> "TensorTranslationLayer":
+ path = Path(path)
+ try:
+ with safe_open(str(path), framework="pt", device="cpu") as handle:
+ metadata = handle.metadata()
+ except SafetensorError as error:
+ raise ValueError("artifact is not a Tensor Translation Layer") from error
+ manifest = json.loads(metadata["manifest"]) if "manifest" in metadata else metadata
+ file_format = manifest.get("format")
+ if file_format == SPLIT_TRANSLATE_FORMAT:
+ if not allow_legacy:
+ raise ValueError("legacy .translate artifact is not an explicit TTL")
+ warnings.warn(
+ ".translate is deprecated. This semantic adapter is classified as TTL.",
+ DeprecationWarning,
+ stacklevel=2,
+ )
+ elif file_format != TTL_FORMAT:
+ raise ValueError("artifact is not a Tensor Translation Layer")
+ if file_format == TTL_FORMAT:
+ if manifest.get("adapter_class") != CompatibilityKind.TTL.value:
+ raise ValueError("TTL adapter class metadata does not match")
+ if manifest.get("support_level") != "semantic/internal":
+ raise ValueError("TTL support level metadata does not match")
+ raw_config = manifest["config"]
+ raw = json.loads(raw_config) if isinstance(raw_config, str) else dict(raw_config)
+ raw["attachment_layers"] = tuple(raw["attachment_layers"])
+ # The neural architecture remains the proven split translator. The
+ # container format, not the in-memory architecture config, is migrated.
+ raw["format"] = SPLIT_TRANSLATE_FORMAT
+ result = cls(SplitTranslateConfig(**raw)).to(device=device, dtype=dtype)
+ state = load_file(str(path), device=str(device))
+ if any(field in name.casefold() for name in state for field in FORBIDDEN_ADAPTER_FIELDS):
+ raise ValueError("TTL contains forbidden model or conversation state")
+ if tensor_checksum(state) != manifest.get("weights_sha256"):
+ raise ValueError("TTL weights checksum does not match")
+ result.load_state_dict({name: value.to(dtype=dtype) for name, value in state.items()})
+ return result
+
+
+@dataclass(frozen=True)
+class LexicalTranslationConfig:
+ model_id: str
+ model_architecture: str
+ model_sha256: str
+ runtime: str
+ runtime_version: str
+ tokenizer_bundle_sha256: str
+ canonical_protocol: str = CANONICAL_P_PROTOCOL
+ format: str = LTL_FORMAT
+ adapter_class: str = CompatibilityKind.LTL.value
+ support_level: str = "lexical/output"
+ control: str = "direct_adaptive_logit_bias"
+ logit_margin: float = 0.01
+ parameter_count: int = 0
+
+ def __post_init__(self) -> None:
+ if self.format != LTL_FORMAT or self.adapter_class != CompatibilityKind.LTL.value:
+ raise ValueError("invalid LTL format or adapter class")
+ if self.support_level != "lexical/output":
+ raise ValueError("invalid LTL support level")
+ if self.canonical_protocol != CANONICAL_P_PROTOCOL:
+ raise ValueError("unsupported canonical P protocol")
+ if self.parameter_count != 0:
+ raise ValueError("the direct adaptive logit-bias LTL has no learned parameters")
+ if self.logit_margin < 0:
+ raise ValueError("logit margin must be non-negative")
+
+
+def _canonical_json(value: object) -> bytes:
+ return json.dumps(
+ value, ensure_ascii=False, sort_keys=True, separators=(",", ":"),
+ ).encode("utf-8")
+
+
+def tokenizer_bundle_checksum(path: str | Path) -> str:
+ """Hash the exact tokenizer and metadata bundle used by the Gemma LTL."""
+ root = Path(path)
+ names = (
+ "chat_template.jinja",
+ "config.json",
+ "generation_config.json",
+ "processor_config.json",
+ "tokenizer_config.json",
+ "tokenizer.json",
+ )
+ digest = hashlib.sha256()
+ for name in names:
+ item = root / name
+ if not item.is_file():
+ raise FileNotFoundError(f"tokenizer bundle file is missing: {item}")
+ digest.update(name.encode("utf-8"))
+ digest.update(b"\0")
+ digest.update(item.read_bytes())
+ return digest.hexdigest()
+
+
+class LexicalTranslationLayer:
+ """Metadata-only lexical/output compatibility for llama.cpp runtimes."""
+
+ adapter_class = CompatibilityKind.LTL.value
+ support_level = "lexical/output"
+ extension = LTL_EXTENSION
+
+ def __init__(self, config: LexicalTranslationConfig) -> None:
+ self.config = config
+
+ def target(self, canonical_value: str, *, route_accepted: bool) -> str | None:
+ """Return an output target only after the universal router accepts it."""
+ if not route_accepted:
+ return None
+ value = str(canonical_value)
+ return value if value else None
+
+ def token_targets(
+ self,
+ canonical_value: str,
+ tokenizer,
+ *,
+ route_accepted: bool,
+ ) -> tuple[int, ...]:
+ target = self.target(canonical_value, route_accepted=route_accepted)
+ if target is None:
+ return ()
+ encoded = tokenizer(target, add_special_tokens=False).input_ids
+ return tuple(int(token_id) for token_id in encoded)
+
+ def adaptive_bias(self, logits: torch.Tensor, target_token_id: int) -> float:
+ """Return the minimum non-negative bias that wins by the configured margin."""
+ flat = logits.detach().float().flatten()
+ if target_token_id < 0 or target_token_id >= flat.numel():
+ raise IndexError("target token is outside the model vocabulary")
+ masked = flat.clone()
+ masked[target_token_id] = -torch.inf
+ required = masked.max() - flat[target_token_id] + self.config.logit_margin
+ return max(0.0, float(required))
+
+ def validate_compatibility(
+ self,
+ *,
+ model_id: str,
+ model_architecture: str,
+ model_sha256: str,
+ runtime: str,
+ runtime_version: str | None = None,
+ tokenizer_bundle_sha256: str | None = None,
+ canonical_protocol: str = CANONICAL_P_PROTOCOL,
+ ) -> None:
+ mismatches = []
+ if model_id != self.config.model_id:
+ mismatches.append("model identifier")
+ if model_architecture != self.config.model_architecture:
+ mismatches.append("model architecture")
+ if model_sha256 != self.config.model_sha256:
+ mismatches.append("model checksum")
+ if runtime != self.config.runtime:
+ mismatches.append("runtime")
+ if runtime_version is not None and runtime_version != self.config.runtime_version:
+ mismatches.append("runtime version")
+ if (
+ tokenizer_bundle_sha256 is not None
+ and tokenizer_bundle_sha256 != self.config.tokenizer_bundle_sha256
+ ):
+ mismatches.append("tokenizer bundle checksum")
+ if canonical_protocol != self.config.canonical_protocol:
+ mismatches.append("canonical protocol")
+ if mismatches:
+ raise ValueError("incompatible LTL: " + ", ".join(mismatches))
+
+ def save(self, path: str | Path) -> None:
+ path = Path(path)
+ if path.suffix != LTL_EXTENSION:
+ raise ValueError(f"LTL artifacts must use the {LTL_EXTENSION} extension")
+ payload = asdict(self.config)
+ payload_bytes = _canonical_json(payload)
+ envelope = {
+ "format": LTL_FORMAT,
+ "payload": payload,
+ "payload_sha256": hashlib.sha256(payload_bytes).hexdigest(),
+ }
+ path.write_bytes(_canonical_json(envelope) + b"\n")
+
+ @classmethod
+ def load(cls, path: str | Path) -> "LexicalTranslationLayer":
+ envelope = json.loads(Path(path).read_text(encoding="utf-8"))
+ if envelope.get("format") != LTL_FORMAT:
+ raise ValueError("artifact is not a Lexical Translation Layer")
+ payload = envelope.get("payload")
+ if not isinstance(payload, dict):
+ raise ValueError("LTL payload is missing")
+ actual = hashlib.sha256(_canonical_json(payload)).hexdigest()
+ if actual != envelope.get("payload_sha256"):
+ raise ValueError("LTL checksum does not match")
+ return cls(LexicalTranslationConfig(**payload))
+
+
+def classify_compatibility_artifact(path: str | Path) -> CompatibilityKind:
+ """Classify modern adapters and the one supported legacy semantic format."""
+ path = Path(path)
+ if path.suffix == LTL_EXTENSION:
+ return CompatibilityKind.LTL
+ try:
+ with safe_open(str(path), framework="pt", device="cpu") as handle:
+ metadata = handle.metadata()
+ manifest = json.loads(metadata["manifest"]) if "manifest" in metadata else metadata
+ file_format = manifest.get("format")
+ except Exception as error:
+ raise ValueError(f"unrecognized compatibility artifact: {path}") from error
+ if file_format in (TTL_FORMAT, SPLIT_TRANSLATE_FORMAT):
+ return CompatibilityKind.TTL
+ raise ValueError(
+ "legacy artifact is research-only and has no active TTL or LTL classification"
+ )
+
+
+def resolve_compatibility(path: str | Path | None) -> CompatibilityResolution:
+ if path is None:
+ return CompatibilityResolution(CompatibilityKind.NATIVE, "native", None)
+ artifact = Path(path)
+ kind = classify_compatibility_artifact(artifact)
+ level = "semantic/internal" if kind is CompatibilityKind.TTL else "lexical/output"
+ return CompatibilityResolution(kind, level, artifact)
diff --git a/src/pcm/planner/interactive_runtimes.py b/src/pcm/planner/interactive_runtimes.py
new file mode 100644
index 0000000000000000000000000000000000000000..ca45c5dc6c24e01be47201634b483ca4cd599fee
--- /dev/null
+++ b/src/pcm/planner/interactive_runtimes.py
@@ -0,0 +1,928 @@
+"""Actual Pythia and llama.cpp runtime adapters for interactive Planner Cache."""
+
+from __future__ import annotations
+
+from dataclasses import dataclass
+import copy
+import hashlib
+import json
+import math
+import os
+from pathlib import Path
+import signal
+import socket
+import subprocess
+import tempfile
+import time
+from typing import Callable
+from urllib.request import Request, urlopen
+
+import torch
+import torch.nn.functional as F
+
+from pcm.planner.canonical import CanonicalPStore
+from pcm.planner.interactive_session import (
+ CanonicalQueryIntent,
+ RELATION_NAMES,
+ file_sha256,
+)
+from pcm.planner.compatibility import (
+ LexicalTranslationLayer,
+ TensorTranslationLayer,
+ tokenizer_bundle_checksum,
+)
+from pcm.planner.personality import merge_active_personality_with_p_cache
+from pcm.planner.pythia_split_translate import PythiaSplitTranslatedModel
+from pcm.planner.split_translator import (
+ ByteEntityEncoder,
+ CanonicalPRouter,
+ FactorizedCanonicalQuery,
+)
+
+
+Emit = Callable[..., object]
+
+
+@dataclass(frozen=True)
+class GenerationResult:
+ text: str
+ latency_seconds: float
+ input_tokens: int | None
+ output_tokens: int | None
+ diagnostics: dict[str, object]
+
+
+def merge_runtime_state(
+ p_cache: CanonicalPStore,
+ personality: CanonicalPStore | None,
+) -> CanonicalPStore:
+ if personality is None:
+ return p_cache
+ merged = merge_active_personality_with_p_cache(p_cache, personality)
+ source_surfaces = getattr(p_cache, "_pcm_value_surfaces", {})
+ if source_surfaces:
+ merged_surfaces = {}
+ for merged_index in merged.valid.nonzero(as_tuple=False).flatten().tolist():
+ for source_index, surface in source_surfaces.items():
+ if (
+ int(merged.entity_id[merged_index]) == int(p_cache.entity_id[source_index])
+ and int(merged.relation_id[merged_index]) == int(p_cache.relation_id[source_index])
+ and int(merged.value_id[merged_index]) == int(p_cache.value_id[source_index])
+ ):
+ merged_surfaces[merged_index] = surface
+ break
+ merged._pcm_value_surfaces = merged_surfaces
+ return merged
+
+
+def _entry(store: CanonicalPStore, index: int) -> dict[str, object]:
+ return {
+ "slot_id": index,
+ "entity": store.cache.labels[index],
+ "entity_id": int(store.entity_id[index]),
+ "relation_id": int(store.relation_id[index]),
+ "relation": (
+ RELATION_NAMES[int(store.relation_id[index])]
+ if 0 <= int(store.relation_id[index]) < len(RELATION_NAMES) else None
+ ),
+ "value_id": int(store.value_id[index]),
+ "metadata_id": int(store.canonical_metadata_id[index]),
+ }
+
+
+def canonical_route(
+ router: CanonicalPRouter,
+ encoder: ByteEntityEncoder,
+ store: CanonicalPStore,
+ query: CanonicalQueryIntent,
+) -> tuple[object | None, list[dict[str, object]]]:
+ if not query.entity or query.relation_id is None or store.cache.occupied == 0:
+ return None, []
+ relation_logits = torch.full((1, router.config.relation_count), -12.0)
+ relation_logits[0, query.relation_id] = 12.0
+ metadata_logits = torch.full((1, router.config.metadata_count), -12.0)
+ metadata_logits[0, 0] = 12.0
+ factorized = FactorizedCanonicalQuery(
+ entity=encoder([query.entity]),
+ relation_logits=relation_logits,
+ metadata_logits=metadata_logits,
+ )
+ index = router.build_index(store, encoder, device="cpu")
+ scores, _features = router.all_scores(factorized, index)
+ candidates = []
+ for slot in torch.argsort(scores[0], descending=True).tolist():
+ score = float(scores[0, slot].detach())
+ if not math.isfinite(score):
+ continue
+ candidates.append({**_entry(store, slot), "score": score})
+ if len(candidates) == 8:
+ break
+ return router.route(factorized, index, top_k=1), candidates
+
+
+class PythiaInteractiveRuntime:
+ name = "pythia"
+ runtime = "transformers-gpt-neox"
+
+ def __init__(
+ self,
+ *,
+ model_path: Path,
+ adapter_path: Path,
+ router_path: Path,
+ max_context_tokens: int = 1024,
+ max_new_tokens: int = 96,
+ temperature: float = 0.7,
+ top_p: float = 0.9,
+ seed: int = 1234,
+ review_model_path: Path | None = None,
+ review_llama_cpp_dir: Path | None = None,
+ review_gpu_layers: int = 0,
+ review_pid_file: Path | None = None,
+ ) -> None:
+ from transformers import AutoModelForCausalLM, AutoTokenizer
+ from transformers.utils import logging as transformers_logging
+
+ if not torch.cuda.is_available():
+ raise RuntimeError("Pythia interactive runtime requires CUDA")
+ transformers_logging.set_verbosity_error()
+ transformers_logging.disable_progress_bar()
+ self.model_path = model_path.resolve()
+ self.adapter_path = adapter_path.resolve()
+ self.router_path = router_path.resolve()
+ self.max_context_tokens = max_context_tokens
+ self.max_new_tokens = max_new_tokens
+ self.temperature = temperature
+ self.top_p = top_p
+ self.seed = seed
+ self.tokenizer = AutoTokenizer.from_pretrained(self.model_path, local_files_only=True)
+ self.tokenizer.pad_token = self.tokenizer.eos_token
+ self.base = AutoModelForCausalLM.from_pretrained(
+ self.model_path, local_files_only=True, dtype=torch.float16,
+ low_cpu_mem_usage=True,
+ ).to("cuda").eval()
+ self.package = TensorTranslationLayer.load(
+ self.adapter_path, device="cuda", dtype=torch.float32,
+ ).eval()
+ self.router = CanonicalPRouter.load(self.router_path, device="cuda").eval()
+ self.encoder = ByteEntityEncoder(128)
+ self.wrapper = PythiaSplitTranslatedModel(
+ self.base, self.package, self.router, self.encoder,
+ ).to("cuda").eval()
+ self.generator = torch.Generator(device="cuda")
+ self.generator.manual_seed(seed)
+ self.review_model_path = (
+ None if review_model_path is None else review_model_path.resolve()
+ )
+ self.review_server = None
+ if self.review_model_path is not None:
+ assert review_llama_cpp_dir is not None
+ self.review_server = LlamaServerProcess(
+ binary=(
+ review_llama_cpp_dir.resolve() / "build/bin/llama-server"
+ ),
+ model=self.review_model_path,
+ control_vector=self.review_model_path.with_suffix(".unused-review.gguf"),
+ attachment_layer=0,
+ context_tokens=2048,
+ gpu_layers=review_gpu_layers,
+ threads=8,
+ pid_file=review_pid_file,
+ startup_timeout=180,
+ )
+
+ @property
+ def model_id(self) -> str:
+ return self.package.config.model_id
+
+ def metadata(self) -> dict[str, object]:
+ import transformers
+
+ return {
+ "model": self.model_id,
+ "model_path": str(self.model_path),
+ "runtime": self.runtime,
+ "runtime_version": transformers.__version__,
+ "compatibility_layer": "ttl",
+ "compatibility_support_level": "semantic/internal",
+ "adapter_artifact": str(self.adapter_path),
+ "adapter_checksum": file_sha256(self.adapter_path),
+ "adapter_format": "planner-cache-ttl-v1",
+ "adapter_attachment_layers": list(self.package.config.attachment_layers),
+ "router_artifact": str(self.router_path),
+ "router_checksum": file_sha256(self.router_path),
+ "router_format": self.router.config.format,
+ "recent_kv_context_tokens": self.max_context_tokens,
+ "generation": {
+ "max_new_tokens": self.max_new_tokens,
+ "temperature": self.temperature,
+ "top_p": self.top_p,
+ "seed": self.seed,
+ },
+ "memory_reviewer": {
+ "mode": (
+ "same-frozen-pythia"
+ if self.review_model_path is None
+ else "separate-frozen-llama-json-reviewer"
+ ),
+ "model_path": (
+ None if self.review_model_path is None
+ else str(self.review_model_path)
+ ),
+ "ttl_or_ltl_active": False,
+ },
+ }
+
+ def _prompt(self, history: list[tuple[str, str]], message: str) -> str:
+ rows = []
+ for user, assistant in history:
+ rows.extend((f"User: {user}", f"Assistant: {assistant}"))
+ rows.extend((f"User: {message}", "Assistant:"))
+ text = "\n".join(rows)
+ tokens = self.tokenizer(text, add_special_tokens=False).input_ids
+ input_limit = max(16, self.max_context_tokens - self.max_new_tokens)
+ if len(tokens) > input_limit:
+ tokens = tokens[-input_limit:]
+ text = self.tokenizer.decode(tokens, skip_special_tokens=True)
+ return text
+
+ def _sample(self, logits: torch.Tensor) -> torch.Tensor:
+ if self.temperature <= 0:
+ return logits.argmax(-1, keepdim=True)
+ probabilities = F.softmax(logits.float() / self.temperature, dim=-1)
+ sorted_probabilities, sorted_indices = probabilities.sort(descending=True)
+ cumulative = sorted_probabilities.cumsum(-1)
+ remove = cumulative - sorted_probabilities > self.top_p
+ sorted_probabilities = sorted_probabilities.masked_fill(remove, 0)
+ sorted_probabilities /= sorted_probabilities.sum(-1, keepdim=True)
+ sampled = torch.multinomial(
+ sorted_probabilities, 1, generator=self.generator,
+ )
+ return sorted_indices.gather(-1, sampled)
+
+ def _emit_step(
+ self,
+ emit: Emit | None,
+ store: CanonicalPStore,
+ query: CanonicalQueryIntent,
+ generation_step: int,
+ ) -> dict[str, object]:
+ diagnostics: dict[str, object] = {
+ "attachment_layers": list(self.package.config.attachment_layers),
+ "query_entity": query.entity,
+ }
+ if not self.wrapper.route_telemetry or not self.wrapper.query_telemetry:
+ if emit:
+ emit(
+ "ROUTER_QUERY", source="model_to_p", entity=query.entity,
+ requested_relation=query.relation, reason=query.reason,
+ generation_step=generation_step,
+ )
+ emit(
+ "ROUTER_CANDIDATES", source="p_cache", candidates=[],
+ generation_step=generation_step,
+ )
+ emit(
+ "ROUTER_REJECT", source="p_cache", reason="no active routable state",
+ entity=query.entity, relation=query.relation,
+ generation_step=generation_step,
+ )
+ emit(
+ "TTL_DISABLE", source="ttl", reason="router did not run",
+ generation_step=generation_step,
+ )
+ diagnostics.update({"router_accepted": False, "gate": 0.0})
+ return diagnostics
+ projected = self.wrapper.query_telemetry[-1]
+ route = self.wrapper.route_telemetry[-1]
+ relation_probability = F.softmax(projected.relation_logits[0, -1].float(), dim=-1)
+ metadata_probability = F.softmax(projected.metadata_logits[0, -1].float(), dim=-1)
+ index = int(route.indices[0, -1, 0])
+ score = float(route.scores[0, -1, 0])
+ accepted = bool(route.accepted[0, -1])
+ selected = _entry(store, index) if route.has_valid else None
+ gate = 0.0
+ if self.wrapper.gate_telemetry:
+ gate = float(self.wrapper.gate_telemetry[-1][0, -1])
+ if emit:
+ emit(
+ "ROUTER_QUERY", source="model_to_p", entity=query.entity,
+ requested_relation=query.relation,
+ projected_relation_probabilities=relation_probability.tolist(),
+ projected_metadata_probabilities=metadata_probability.tolist(),
+ entity_anchor="tokenizer-independent-byte-anchor" if query.entity else "learned-hidden-query",
+ generation_step=generation_step,
+ )
+ emit(
+ "ROUTER_CANDIDATES", source="p_cache",
+ candidates=[] if selected is None else [{**selected, "score": score}],
+ generation_step=generation_step,
+ )
+ emit(
+ "ROUTER_ACCEPT" if accepted else "ROUTER_REJECT",
+ source="p_cache", score=score, selected_state=selected,
+ generation_step=generation_step,
+ )
+ emit(
+ "TTL_ENABLE" if accepted and gate > 0 else "TTL_DISABLE",
+ source="ttl", gate=gate,
+ attachment_layers=list(self.package.config.attachment_layers),
+ selected_state=selected,
+ generation_step=generation_step,
+ )
+ emit(
+ "TTL_OUTPUT", source="ttl", gate=gate,
+ active=accepted and gate > 0, selected_state=selected,
+ generation_step=generation_step,
+ )
+ diagnostics.update({
+ "router_accepted": accepted,
+ "router_score": score,
+ "selected_state": selected,
+ "gate": gate,
+ "projected_relation_probabilities": relation_probability.tolist(),
+ })
+ return diagnostics
+
+ def generate(
+ self,
+ message: str,
+ history: list[tuple[str, str]],
+ p_cache: CanonicalPStore,
+ personality: CanonicalPStore | None,
+ query: CanonicalQueryIntent,
+ *,
+ emit: Emit | None = None,
+ raw_messages: list[dict[str, object]] | None = None,
+ ) -> GenerationResult:
+ prompt = self._prompt(history, message)
+ encoded = self.tokenizer(prompt, return_tensors="pt", add_special_tokens=False)
+ input_ids = encoded.input_ids.to("cuda")
+ attention_mask = encoded.attention_mask.to("cuda")
+ input_count = int(input_ids.shape[1])
+ active_store = merge_runtime_state(p_cache, personality)
+ use_store = active_store if query.entity is not None else None
+ generated: list[int] = []
+ past = None
+ started = time.perf_counter()
+ diagnostics: dict[str, object] = {
+ "generation_steps": [],
+ "recent_context_turns": len(history),
+ "recent_context_characters": len(prompt),
+ "recent_context_input_tokens": input_count,
+ }
+ with torch.inference_mode():
+ for step in range(self.max_new_tokens):
+ current = input_ids if past is None else input_ids[:, -1:]
+ output = self.wrapper(
+ input_ids=current,
+ attention_mask=attention_mask,
+ past_key_values=past,
+ p_store=use_store,
+ query_entity_surfaces=[query.entity] if query.entity else None,
+ collect_telemetry=emit is not None,
+ use_cache=True,
+ )
+ step_diagnostics = self._emit_step(
+ emit, active_store, query, generation_step=step,
+ )
+ diagnostics["generation_steps"].append(step_diagnostics)
+ next_token = self._sample(output.logits[:, -1])
+ token_id = int(next_token[0, 0])
+ generated.append(token_id)
+ past = output.past_key_values
+ input_ids = torch.cat((input_ids, next_token), dim=1)
+ attention_mask = torch.cat((attention_mask, torch.ones_like(next_token)), dim=1)
+ if token_id == self.tokenizer.eos_token_id:
+ break
+ decoded = self.tokenizer.decode(generated, skip_special_tokens=True)
+ if "\nUser:" in decoded or "\nuser:" in decoded:
+ break
+ text = self.tokenizer.decode(generated, skip_special_tokens=True)
+ text = re_split_user(text).strip()
+ return GenerationResult(
+ text=text,
+ latency_seconds=time.perf_counter() - started,
+ input_tokens=input_count,
+ output_tokens=len(generated),
+ diagnostics=diagnostics,
+ )
+
+ def review_memory(
+ self, prompt: str, schema: dict[str, object],
+ ) -> str:
+ """Generate hidden review JSON with the frozen base and no TTL path."""
+ if self.review_server is not None:
+ self.review_server.ensure(0.0, vector_key=None)
+ body = json.dumps({
+ "messages": [
+ {
+ "role": "system",
+ "content": "Return only valid JSON for the supplied schema.",
+ },
+ {"role": "user", "content": prompt},
+ ],
+ "max_tokens": 256,
+ "temperature": 0.0,
+ "top_p": 1.0,
+ "seed": self.seed,
+ "stream": False,
+ "response_format": {"type": "json_schema", "schema": schema},
+ }, ensure_ascii=False).encode("utf-8")
+ request = Request(
+ f"http://127.0.0.1:{self.review_server.port}/v1/chat/completions",
+ data=body, headers={"Content-Type": "application/json"},
+ )
+ with urlopen(request, timeout=600) as response:
+ result = json.load(response)
+ return str(result["choices"][0]["message"]["content"])
+ del schema # Transformers generation has no native JSON grammar.
+ review_text = (
+ "Complete each memory-review task with JSON only.\n"
+ "Example task: user says haha okay.\n"
+ "JSON: {\"operations\":[{\"op\":\"IGNORE\",\"entity\":null,"
+ "\"relation\":null,\"value\":null,\"confidence\":1.0,"
+ "\"source\":\"explicit_user\"}]}\n"
+ "Example task: user says I put the key in the drawer.\n"
+ "JSON: {\"operations\":[{\"op\":\"CREATE\",\"entity\":\"key\","
+ "\"relation\":\"location\",\"value\":\"drawer\","
+ "\"confidence\":1.0,\"source\":\"rp_action\"}]}\n"
+ f"Memory-review task:\n{prompt}\nJSON:"
+ )
+ encoded = self.tokenizer(
+ review_text, return_tensors="pt", add_special_tokens=False,
+ )
+ input_limit = self.max_context_tokens - 160
+ input_ids = encoded.input_ids
+ if input_ids.shape[1] > input_limit:
+ prefix_tokens = min(240, input_limit // 3)
+ input_ids = torch.cat((
+ input_ids[:, :prefix_tokens],
+ input_ids[:, -(input_limit - prefix_tokens):],
+ ), dim=1)
+ input_ids = input_ids.to("cuda")
+ attention_mask = torch.ones_like(input_ids)
+ with torch.inference_mode():
+ output = self.base.generate(
+ input_ids=input_ids,
+ attention_mask=attention_mask,
+ max_new_tokens=160,
+ do_sample=False,
+ use_cache=True,
+ pad_token_id=self.tokenizer.eos_token_id,
+ eos_token_id=self.tokenizer.eos_token_id,
+ )
+ return self.tokenizer.decode(
+ output[0, input_ids.shape[1]:], skip_special_tokens=True,
+ )
+
+ def close(self) -> None:
+ if self.review_server is not None:
+ self.review_server.stop()
+ self.wrapper.close()
+
+
+def re_split_user(text: str) -> str:
+ for marker in ("\nUser:", "\nuser:"):
+ if marker in text:
+ return text.split(marker, 1)[0]
+ return text
+
+
+def gemma_chat_request_body(
+ messages: list[dict[str, object]],
+ *,
+ max_tokens: int,
+ temperature: float,
+ top_p: float,
+ seed: int,
+ target_token_ids: list[int] | tuple[int, ...] = (),
+) -> dict[str, object]:
+ """Build a native request without altering the chat message structure."""
+ body: dict[str, object] = {
+ "messages": copy.deepcopy(messages),
+ "max_tokens": max_tokens,
+ "temperature": temperature,
+ "top_p": top_p,
+ "seed": seed,
+ "stream": False,
+ }
+ if target_token_ids:
+ body["logit_bias"] = {
+ str(token_id): 100.0 for token_id in target_token_ids
+ }
+ return body
+
+
+class LlamaServerProcess:
+ """Managed llama-server that can switch inert and control-vector modes."""
+
+ def __init__(
+ self,
+ *,
+ binary: Path,
+ model: Path,
+ control_vector: Path,
+ attachment_layer: int,
+ context_tokens: int,
+ gpu_layers: int,
+ threads: int,
+ startup_timeout: float = 90.0,
+ pid_file: Path | None = None,
+ popen_factory=subprocess.Popen,
+ ) -> None:
+ self.binary = binary
+ self.model = model
+ self.control_vector = control_vector
+ self.attachment_layer = attachment_layer
+ self.context_tokens = context_tokens
+ self.gpu_layers = gpu_layers
+ self.threads = threads
+ self.startup_timeout = startup_timeout
+ self.pid_file = pid_file
+ self.popen_factory = popen_factory
+ self.process = None
+ self.port: int | None = None
+ self.strength: float | None = None
+ self.vector_key: str | None = None
+
+ def command(self, port: int, strength: float) -> list[str]:
+ command = [
+ str(self.binary), "--model", str(self.model),
+ "--host", "127.0.0.1", "--port", str(port),
+ "--ctx-size", str(self.context_tokens), "--parallel", "1",
+ "--threads", str(self.threads), "--threads-batch", str(self.threads),
+ "--gpu-layers", str(self.gpu_layers), "--no-warmup",
+ "--no-context-shift", "--log-disable", "--reasoning", "off",
+ ]
+ if strength != 0:
+ command.extend((
+ "--control-vector-scaled", f"{self.control_vector}:{strength}",
+ "--control-vector-layer-range", str(self.attachment_layer),
+ str(self.attachment_layer),
+ ))
+ return command
+
+ def ensure(self, strength: float, *, vector_key: str | None = None) -> dict[str, object]:
+ if (
+ self.process is not None
+ and self.process.poll() is None
+ and self.strength == strength
+ and self.vector_key == vector_key
+ ):
+ return {"restarted": False, "port": self.port, "strength": strength}
+ self.stop()
+ with socket.socket() as reservation:
+ reservation.bind(("127.0.0.1", 0))
+ port = reservation.getsockname()[1]
+ command = self.command(port, strength)
+ self.process = self.popen_factory(
+ command, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
+ start_new_session=True,
+ )
+ self.port = port
+ self.strength = strength
+ self.vector_key = vector_key
+ if self.pid_file is not None:
+ self.pid_file.write_text(str(self.process.pid) + "\n")
+ deadline = time.monotonic() + self.startup_timeout
+ while True:
+ if self.process.poll() is not None:
+ self.process = None
+ if self.pid_file is not None:
+ self.pid_file.unlink(missing_ok=True)
+ raise RuntimeError("llama-server exited before becoming ready")
+ try:
+ with urlopen(f"http://127.0.0.1:{port}/health", timeout=1):
+ break
+ except Exception:
+ if time.monotonic() >= deadline:
+ self.stop()
+ raise TimeoutError("llama-server did not become ready")
+ time.sleep(0.25)
+ return {"restarted": True, "port": port, "strength": strength, "command": command}
+
+ def stop(self) -> None:
+ process = self.process
+ self.process = None
+ self.port = None
+ self.strength = None
+ self.vector_key = None
+ if self.pid_file is not None:
+ self.pid_file.unlink(missing_ok=True)
+ if process is None or process.poll() is not None:
+ return
+ try:
+ os.killpg(process.pid, signal.SIGTERM)
+ except ProcessLookupError:
+ return
+ try:
+ process.wait(timeout=15)
+ except subprocess.TimeoutExpired:
+ try:
+ os.killpg(process.pid, signal.SIGKILL)
+ except ProcessLookupError:
+ pass
+ process.wait(timeout=15)
+
+
+class GemmaInteractiveRuntime:
+ name = "gemma"
+ runtime = "llama.cpp-server-gguf"
+
+ def __init__(
+ self,
+ *,
+ model_path: Path,
+ adapter_path: Path,
+ router_path: Path,
+ llama_cpp_dir: Path,
+ tokenizer_bundle: Path,
+ max_context_tokens: int = 4096,
+ max_new_tokens: int = 128,
+ temperature: float = 0.7,
+ top_p: float = 0.9,
+ seed: int = 1234,
+ gpu_layers: int = 12,
+ threads: int = 8,
+ pid_file: Path | None = None,
+ ) -> None:
+ self.model_path = model_path.resolve()
+ self.adapter_path = adapter_path.resolve()
+ self.router_path = router_path.resolve()
+ self.llama_cpp_dir = llama_cpp_dir.resolve()
+ self.tokenizer_bundle = tokenizer_bundle.resolve()
+ self.server_binary = self.llama_cpp_dir / "build/bin/llama-server"
+ self.cli_binary = self.llama_cpp_dir / "build/bin/llama-cli"
+ self.supports_open_values = True
+ self.package = LexicalTranslationLayer.load(self.adapter_path)
+ from transformers import AutoTokenizer
+ self.tokenizer = AutoTokenizer.from_pretrained(
+ self.tokenizer_bundle, local_files_only=True,
+ )
+ self.tokenizer_bundle_sha256 = tokenizer_bundle_checksum(self.tokenizer_bundle)
+ self.router = CanonicalPRouter.load(self.router_path)
+ self.encoder = ByteEntityEncoder(128)
+ self.max_context_tokens = max_context_tokens
+ self.max_new_tokens = max_new_tokens
+ self.temperature = temperature
+ self.top_p = top_p
+ self.seed = seed
+ self.model_sha256 = file_sha256(self.model_path)
+ version = subprocess.run(
+ [str(self.cli_binary), "--version"], check=True, text=True,
+ capture_output=True,
+ )
+ self.llama_version = "\n".join(
+ part.strip() for part in (version.stdout, version.stderr) if part.strip()
+ )
+ self.package.validate_compatibility(
+ model_id=self.package.config.model_id,
+ model_architecture=self.package.config.model_architecture,
+ model_sha256=self.model_sha256,
+ runtime="llama.cpp",
+ runtime_version=self.llama_version,
+ tokenizer_bundle_sha256=self.tokenizer_bundle_sha256,
+ )
+ self._temporary = tempfile.TemporaryDirectory(prefix="planner-cache-gemma-chat-")
+ self.control_vector = Path(self._temporary.name) / "unused-ltl-control-vector.gguf"
+ self.server = LlamaServerProcess(
+ binary=self.server_binary, model=self.model_path,
+ control_vector=self.control_vector,
+ attachment_layer=0,
+ context_tokens=max_context_tokens, gpu_layers=gpu_layers, threads=threads,
+ pid_file=pid_file,
+ )
+
+ @property
+ def model_id(self) -> str:
+ return self.package.config.model_id
+
+ def metadata(self) -> dict[str, object]:
+ return {
+ "model": self.model_id,
+ "model_path": str(self.model_path),
+ "model_gguf_sha256": self.model_sha256,
+ "runtime": self.runtime,
+ "runtime_version": self.llama_version,
+ "llama_cpp_dir": str(self.llama_cpp_dir),
+ "compatibility_layer": "ltl",
+ "compatibility_support_level": "lexical/output",
+ "adapter_artifact": str(self.adapter_path),
+ "adapter_checksum": file_sha256(self.adapter_path),
+ "adapter_format": self.package.config.format,
+ "adapter_attachment_layers": [],
+ "lexical_control": self.package.config.control,
+ "tokenizer_bundle": str(self.tokenizer_bundle),
+ "tokenizer_bundle_sha256": self.tokenizer_bundle_sha256,
+ "adapter_parameter_count": self.package.config.parameter_count,
+ "router_artifact": str(self.router_path),
+ "router_checksum": file_sha256(self.router_path),
+ "router_format": self.router.config.format,
+ "recent_kv_context_tokens": self.max_context_tokens,
+ "generation": {
+ "max_new_tokens": self.max_new_tokens,
+ "temperature": self.temperature,
+ "top_p": self.top_p,
+ "seed": self.seed,
+ },
+ "memory_reviewer": {
+ "mode": "same-frozen-gemma",
+ "model_path": str(self.model_path),
+ "ttl_or_ltl_active": False,
+ },
+ }
+
+ def _route_and_translate(
+ self,
+ store: CanonicalPStore,
+ query: CanonicalQueryIntent,
+ emit: Emit | None,
+ ) -> tuple[str | None, dict[str, object]]:
+ if emit:
+ emit(
+ "ROUTER_QUERY", source="canonical_query", entity=query.entity,
+ relation=query.relation, reason=query.reason,
+ )
+ route, candidates = canonical_route(self.router, self.encoder, store, query)
+ if emit:
+ emit("ROUTER_CANDIDATES", source="p_cache", candidates=candidates)
+ if route is None or not route.has_valid:
+ if emit:
+ emit("ROUTER_REJECT", source="p_cache", reason="incomplete query or empty state")
+ emit("LTL_DISABLE", source="ltl", gate=0.0, runtime_inert=True)
+ return None, {"router_accepted": False, "gate": 0.0, "runtime_inert": True}
+ index = int(route.indices[0, 0])
+ score = float(route.scores[0, 0].detach())
+ accepted = bool(route.accepted[0])
+ selected = _entry(store, index)
+ if emit:
+ emit(
+ "ROUTER_ACCEPT" if accepted else "ROUTER_REJECT",
+ source="p_cache", score=score, selected_state=selected,
+ )
+ value_surface = getattr(store, "_pcm_value_surfaces", {}).get(index)
+ target = self.package.target(str(value_surface or ""), route_accepted=accepted)
+ active = target is not None
+ gate = 1.0 if active else 0.0
+ if emit:
+ emit(
+ "LTL_ENABLE" if active else "LTL_DISABLE",
+ source="ltl", gate=gate,
+ selected_state=selected, value_surface=value_surface,
+ lexical_control=self.package.config.control, runtime_inert=not active,
+ )
+ token_targets = self.package.token_targets(
+ str(value_surface or ""), self.tokenizer, route_accepted=accepted,
+ )
+ if emit and token_targets:
+ emit(
+ "LTL_TOKEN_TARGET", source="ltl", target_text=target,
+ target_token_ids=list(token_targets),
+ strategy=self.package.config.control,
+ model_gguf_sha256=self.model_sha256,
+ llama_cpp_version=self.llama_version,
+ )
+ return target, {
+ "router_accepted": accepted,
+ "router_score": score,
+ "selected_state": selected,
+ "gate": gate,
+ "runtime_inert": not active,
+ "value_surface": value_surface,
+ "compatibility_layer": "ltl",
+ "lexical_target": target,
+ "lexical_target_token_ids": list(token_targets),
+ }
+
+ def generate(
+ self,
+ message: str,
+ history: list[tuple[str, str]],
+ p_cache: CanonicalPStore,
+ personality: CanonicalPStore | None,
+ query: CanonicalQueryIntent,
+ *,
+ emit: Emit | None = None,
+ raw_messages: list[dict[str, object]] | None = None,
+ ) -> GenerationResult:
+ active_store = merge_runtime_state(p_cache, personality)
+ target, diagnostics = self._route_and_translate(active_store, query, emit)
+ server_started = time.perf_counter()
+ server_state = self.server.ensure(0.0, vector_key=None)
+ diagnostics["server_restart_latency_seconds"] = time.perf_counter() - server_started
+ diagnostics["server_restarted"] = server_state["restarted"]
+ if raw_messages is not None:
+ # Browser messages remain an opaque native llama.cpp input. Planner
+ # Cache observes the newest user text separately and never rewrites
+ # this structure for memory, routing, or compatibility handling.
+ messages = copy.deepcopy(raw_messages)
+ diagnostics["message_path"] = "native-structured-passthrough"
+ diagnostics["recent_context_turns"] = sum(
+ 1 for row in messages if row.get("role") == "user"
+ )
+ diagnostics["native_message_sha256"] = hashlib.sha256(
+ json.dumps(
+ messages, ensure_ascii=False, sort_keys=True,
+ separators=(",", ":"),
+ ).encode("utf-8")
+ ).hexdigest()
+ else:
+ # The legacy terminal front end has no structured message array.
+ current = {"role": "user", "content": message}
+ character_budget = max(
+ 512, (self.max_context_tokens - self.max_new_tokens) * 3
+ )
+ retained: list[tuple[str, str]] = []
+ used = len(message)
+ for user, assistant in reversed(history):
+ cost = len(user) + len(assistant) + 32
+ if used + cost > character_budget:
+ break
+ retained.append((user, assistant))
+ used += cost
+ messages = []
+ for user, assistant in reversed(retained):
+ messages.extend((
+ {"role": "user", "content": user},
+ {"role": "assistant", "content": assistant},
+ ))
+ messages.append(current)
+ diagnostics["message_path"] = "terminal-history"
+ diagnostics["recent_context_turns"] = len(retained)
+ diagnostics["recent_context_characters"] = used
+ target_ids = diagnostics.get("lexical_target_token_ids", [])
+ request_body = gemma_chat_request_body(
+ messages,
+ max_tokens=self.max_new_tokens,
+ temperature=self.temperature,
+ top_p=self.top_p,
+ seed=self.seed,
+ target_token_ids=target_ids,
+ )
+ body = json.dumps(request_body).encode()
+ started = time.perf_counter()
+ request = Request(
+ f"http://127.0.0.1:{self.server.port}/v1/chat/completions",
+ data=body, headers={"Content-Type": "application/json"},
+ )
+ with urlopen(request, timeout=300) as response:
+ result = json.load(response)
+ usage = result.get("usage", {})
+ # Preserve llama.cpp's assistant content exactly. Trimming here would
+ # alter the assistant message that the Web UI returns on the next turn.
+ text = str(result["choices"][0]["message"]["content"])
+ if emit and target is not None:
+ observed = target.casefold() in text.casefold()
+ if observed:
+ emit("LTL_COMPLETE", source="ltl", target_text=target)
+ else:
+ emit(
+ "LTL_DISABLE", source="ltl", target_text=target,
+ reason="target sequence was not completed",
+ )
+ return GenerationResult(
+ text=text,
+ latency_seconds=time.perf_counter() - started,
+ input_tokens=usage.get("prompt_tokens"),
+ output_tokens=usage.get("completion_tokens"),
+ diagnostics=diagnostics,
+ )
+
+ def review_memory(
+ self, prompt: str, schema: dict[str, object],
+ ) -> str:
+ """Run a separate grammar-constrained review without LTL controls."""
+ self.server.ensure(0.0, vector_key=None)
+ body = json.dumps({
+ "messages": [
+ {
+ "role": "system",
+ "content": (
+ "Return only valid JSON for the supplied schema. "
+ "This is a hidden memory review, not a user-visible reply."
+ ),
+ },
+ {"role": "user", "content": prompt},
+ ],
+ "max_tokens": 256,
+ "temperature": 0.0,
+ "top_p": 1.0,
+ "seed": self.seed,
+ "stream": False,
+ "response_format": {
+ "type": "json_schema",
+ "schema": schema,
+ },
+ }, ensure_ascii=False).encode("utf-8")
+ request = Request(
+ f"http://127.0.0.1:{self.server.port}/v1/chat/completions",
+ data=body, headers={"Content-Type": "application/json"},
+ )
+ with urlopen(request, timeout=300) as response:
+ result = json.load(response)
+ return str(result["choices"][0]["message"]["content"])
+
+ def close(self) -> None:
+ self.server.stop()
+ self._temporary.cleanup()
diff --git a/src/pcm/planner/interactive_session.py b/src/pcm/planner/interactive_session.py
new file mode 100644
index 0000000000000000000000000000000000000000..8f35a750192c88efdf87023e03e27de81977280a
--- /dev/null
+++ b/src/pcm/planner/interactive_session.py
@@ -0,0 +1,760 @@
+"""Shared interactive session state, event recording, and conservative extraction."""
+
+from __future__ import annotations
+
+from collections import Counter, deque
+from dataclasses import asdict, dataclass
+from datetime import datetime, timezone
+import hashlib
+import json
+from pathlib import Path
+import re
+import subprocess
+import time
+from typing import Callable, Iterable
+
+import torch
+
+from pcm.planner.cache import (
+ Freshness,
+ Persistence,
+ SlotSource,
+ SlotType,
+ StateOperation,
+)
+from pcm.planner.canonical import (
+ CANONICAL_P_PROTOCOL,
+ CANONICAL_VALUE_LABELS,
+ CanonicalPConfig,
+ CanonicalPStore,
+)
+from pcm.planner.personality import (
+ EvidenceAuthority,
+ EvidenceRecord,
+ FactorizedPersonalityCanonicalizer,
+ PersonalityPackage,
+ PersonalityQuery,
+ PersonalityRouter,
+ PersonalityStatus,
+ PersonalityType,
+)
+from pcm.planner.representation import CANONICAL, FactorizedStateRepresentation
+
+
+EventSink = Callable[[str], None]
+RELATION_NAMES = ("owner", "location", "status")
+VALUE_LOOKUP = {
+ label.casefold(): (index, label)
+ for index, label in enumerate(CANONICAL_VALUE_LABELS)
+}
+
+
+def utc_now() -> str:
+ return datetime.now(timezone.utc).isoformat(timespec="microseconds")
+
+
+def file_sha256(path: str | Path) -> str:
+ digest = hashlib.sha256()
+ with Path(path).open("rb") as handle:
+ for block in iter(lambda: handle.read(16 * 1024 * 1024), b""):
+ digest.update(block)
+ return digest.hexdigest()
+
+
+def git_commit(root: Path) -> str:
+ completed = subprocess.run(
+ ["git", "rev-parse", "HEAD"], cwd=root, text=True,
+ capture_output=True, check=False,
+ )
+ return completed.stdout.strip() if completed.returncode == 0 else "unavailable"
+
+
+def safe_session_stamp(stamp: str) -> str:
+ return stamp.replace(":", "").replace("+", "p").replace(".", "-")
+
+
+class SessionRecorder:
+ """Append-only JSONL recorder with a human-readable final snapshot."""
+
+ def __init__(
+ self,
+ session_root: str | Path,
+ runtime_name: str,
+ metadata: dict[str, object],
+ *,
+ enabled: bool = True,
+ started_at: str | None = None,
+ ) -> None:
+ self.enabled = enabled
+ self.started_at = started_at or utc_now()
+ self.started_monotonic = time.monotonic()
+ self.runtime_name = runtime_name
+ self.turn = 0
+ self.event_counts: Counter[str] = Counter()
+ self.recent_events: deque[dict[str, object]] = deque(maxlen=50)
+ self.metadata = dict(metadata)
+ self.metadata.update({
+ "session_id": f"{safe_session_stamp(self.started_at)}-{runtime_name}",
+ "start_timestamp": self.started_at,
+ })
+ root = Path(session_root)
+ candidate = root / str(self.metadata["session_id"])
+ suffix = 1
+ while candidate.exists():
+ candidate = root / f"{self.metadata['session_id']}-{suffix}"
+ suffix += 1
+ self.directory = candidate
+ self.transcript_path = candidate / "transcript.jsonl"
+ self.events_path = candidate / "events.jsonl"
+ self.session_path = candidate / "session.json"
+ self.final_state_path = candidate / "final-state.json"
+ self._transcript_handle = None
+ self._events_handle = None
+ self._closed = False
+ if enabled:
+ candidate.mkdir(parents=True, exist_ok=False)
+ self._transcript_handle = self.transcript_path.open("a", encoding="utf-8")
+ self._events_handle = self.events_path.open("a", encoding="utf-8")
+ self.session_path.write_text(
+ json.dumps(self.metadata, indent=2, sort_keys=True) + "\n",
+ encoding="utf-8",
+ )
+ self.event("SESSION_START", source="session", metadata=self.metadata)
+
+ def event(self, event: str, *, turn: int | None = None, **fields: object) -> dict[str, object]:
+ row = {
+ "timestamp": utc_now(),
+ "turn": self.turn if turn is None else turn,
+ "event": event,
+ **fields,
+ }
+ self.event_counts[event] += 1
+ self.recent_events.append(row)
+ if self.enabled and self._events_handle is not None:
+ self._events_handle.write(json.dumps(row, sort_keys=True, ensure_ascii=False) + "\n")
+ self._events_handle.flush()
+ return row
+
+ def transcript(
+ self,
+ *,
+ role: str,
+ text: str,
+ model: str,
+ runtime: str,
+ latency_seconds: float | None = None,
+ input_tokens: int | None = None,
+ output_tokens: int | None = None,
+ ) -> None:
+ row = {
+ "timestamp": utc_now(),
+ "turn": self.turn,
+ "role": role,
+ "text": text,
+ "raw_user_text": text if role == "user" else None,
+ "raw_model_output": text if role == "assistant" else None,
+ "model": model,
+ "runtime": runtime,
+ "generation_latency_seconds": latency_seconds,
+ "input_tokens": input_tokens,
+ "output_tokens": output_tokens,
+ }
+ if self.enabled and self._transcript_handle is not None:
+ self._transcript_handle.write(
+ json.dumps(row, sort_keys=True, ensure_ascii=False) + "\n"
+ )
+ self._transcript_handle.flush()
+
+ def save_event(self, reason: str, **fields: object) -> None:
+ self.event("SESSION_SAVE", source="session", reason=reason, **fields)
+
+ def finalize(self, state: dict[str, object], *, reason: str) -> None:
+ if self._closed:
+ return
+ elapsed = time.monotonic() - self.started_monotonic
+ self.event("SESSION_END", source="session", reason=reason, runtime_seconds=elapsed)
+ final = {
+ **state,
+ "session": self.metadata,
+ "event_counts": dict(sorted(self.event_counts.items())),
+ "total_turns": self.turn,
+ "total_runtime_seconds": elapsed,
+ "end_reason": reason,
+ "end_timestamp": utc_now(),
+ }
+ if self.enabled:
+ self.final_state_path.write_text(
+ json.dumps(final, indent=2, sort_keys=True, ensure_ascii=False) + "\n",
+ encoding="utf-8",
+ )
+ for handle in (self._transcript_handle, self._events_handle):
+ if handle is not None:
+ handle.close()
+ self._closed = True
+
+
+@dataclass(frozen=True)
+class CanonicalQueryIntent:
+ entity: str | None
+ relation_id: int | None
+ relation: str | None
+ reason: str
+
+
+@dataclass(frozen=True)
+class MutationIntent:
+ action: str
+ entity: str
+ relation_id: int
+ value_id: int | None = None
+ value: str | None = None
+ translator_compatible: bool = True
+
+
+class CanonicalStateManager:
+ """Conservative text-to-existing-canonical-state interface.
+
+ It recognizes only explicit current-state statements and invalidations. It
+ never asks the language model to infer state and never extends the fixed
+ canonical value vocabulary.
+ """
+
+ def __init__(
+ self,
+ representation: FactorizedStateRepresentation,
+ *,
+ slots: int = 128,
+ store: CanonicalPStore | None = None,
+ ) -> None:
+ self.representation = representation.cpu().eval()
+ self.store = store or CanonicalPStore(CanonicalPConfig(
+ slots=slots, width=512, dtype=torch.float32, device="cpu"
+ ))
+ self.surface_values: dict[int, str] = {}
+ self.translator_compatible: dict[int, bool] = {}
+ for index in self.store.valid.nonzero(as_tuple=False).flatten().tolist():
+ value_id = int(self.store.value_id[index])
+ if 0 <= value_id < len(CANONICAL_VALUE_LABELS):
+ self.surface_values[index] = CANONICAL_VALUE_LABELS[value_id]
+ self.translator_compatible[index] = True
+
+ @staticmethod
+ def entity_id(surface: str) -> int:
+ return int.from_bytes(
+ hashlib.sha256(surface.casefold().encode("utf-8")).digest()[:8], "big"
+ ) & ((1 << 63) - 1)
+
+ def vector(self, entity: str, relation_id: int, value_id: int) -> torch.Tensor:
+ proof_entities = {"silver key": 0, "gold key": 1}
+ factor_entity_id = proof_entities.get(
+ entity.casefold(), self.entity_id(entity) % 24,
+ )
+ with torch.inference_mode():
+ return self.representation.encode(
+ torch.tensor([factor_entity_id]),
+ torch.tensor([relation_id]),
+ torch.tensor([value_id]),
+ torch.tensor([CANONICAL]),
+ )[0].float()
+
+ @staticmethod
+ def _clean_entity(value: str) -> str:
+ value = re.sub(
+ r"^(?:please\s+)?(?:remember|note)(?:\s+that)?\s+",
+ "", value, flags=re.IGNORECASE,
+ )
+ value = re.sub(r"^the\s+", "", value, flags=re.IGNORECASE)
+ return value.strip(" \t\n\r.,:;!?\"'")[:160]
+
+ @staticmethod
+ def _value_pattern() -> str:
+ labels = sorted(CANONICAL_VALUE_LABELS, key=len, reverse=True)
+ return "(?:" + "|".join(re.escape(label) for label in labels) + ")"
+
+ @staticmethod
+ def _canonical_value(value: str) -> tuple[int, str, bool]:
+ cleaned = value.strip(" \t\n\r.,:;!?\"'")[:160]
+ known = VALUE_LOOKUP.get(cleaned.casefold())
+ if known is not None:
+ return known[0], known[1], True
+ value_id = int.from_bytes(
+ hashlib.sha256(cleaned.casefold().encode("utf-8")).digest()[:8], "big"
+ ) % len(CANONICAL_VALUE_LABELS)
+ return value_id, cleaned, False
+
+ def extract_mutations(self, text: str) -> list[MutationIntent]:
+ value = self._value_pattern()
+ flags = re.IGNORECASE
+ patterns = (
+ (0, rf"^(?P.+?)\s+(?:currently\s+)?(?:belongs\s+to|is\s+owned\s+by)\s+(?P{value})[.!]?$"),
+ (0, rf"^(?P{value})\s+(?:currently\s+)?owns\s+(?P.+?)[.!]?$"),
+ (1, rf"^(?P.+?)\s+(?:is\s+located\s+in|is\s+located\s+at|is\s+currently\s+in)\s+(?P{value})[.!]?$"),
+ (2, rf"^(?:the\s+)?(?:current\s+)?status\s+of\s+(?P.+?)\s+is\s+(?P{value})[.!]?$"),
+ (0, rf"^remember:\s*(?P.+?)\.owner\s*=\s*(?P{value})\s*$"),
+ (1, rf"^remember:\s*(?P.+?)\.location\s*=\s*(?P{value})\s*$"),
+ (2, rf"^remember:\s*(?P.+?)\.status\s*=\s*(?P{value})\s*$"),
+ )
+ for relation_id, pattern in patterns:
+ match = re.match(pattern, text.strip(), flags)
+ if match:
+ raw_value = match.group("value")
+ value_id, canonical_label = VALUE_LOOKUP[raw_value.casefold()]
+ return [MutationIntent(
+ "upsert", self._clean_entity(match.group("entity")),
+ relation_id, value_id, canonical_label, True,
+ )]
+ free_patterns = (
+ (
+ 1,
+ r"^(?:my\s+character|i|we|[\w'-]+)\s+"
+ r"(?:left|placed|put|set)\s+(?:the\s+)?(?P.+?)\s+"
+ r"(?:on|in|at|inside|beside|under)\s+(?:the\s+)?(?P[^.!?]+)[.!]?$",
+ ),
+ (
+ 0,
+ r"^(?:please\s+)?(?:remember|note)(?:\s+that)?\s+"
+ r"(?:the\s+)?(?P.+?)\s+(?:belongs\s+to|is\s+owned\s+by)\s+"
+ r"(?P[^.!?]+)[.!]?$",
+ ),
+ (
+ 1,
+ r"^(?:please\s+)?(?:remember|note)(?:\s+that)?\s+"
+ r"(?:the\s+)?(?P.+?)\s+"
+ r"(?:is\s+on|is\s+in|is\s+at|is\s+inside|is\s+beside|is\s+under)\s+"
+ r"(?:the\s+)?(?P[^.!?]+)[.!]?$",
+ ),
+ )
+ for relation_id, pattern in free_patterns:
+ match = re.match(pattern, text.strip(), flags)
+ if match:
+ value_id, surface, compatible = self._canonical_value(match.group("value"))
+ return [MutationIntent(
+ "upsert", self._clean_entity(match.group("entity")),
+ relation_id, value_id, surface, compatible,
+ )]
+ invalidate = re.match(
+ r"^(?:forget|invalidate):?\s*(?P.+?)(?:\.|\s+)(?Powner|location|status)[.!]?$",
+ text.strip(), flags,
+ )
+ if invalidate:
+ relation = invalidate.group("relation").casefold()
+ return [MutationIntent(
+ "invalidate", self._clean_entity(invalidate.group("entity")),
+ RELATION_NAMES.index(relation),
+ )]
+ return []
+
+ def extract_manual_mutations(self, text: str) -> list[MutationIntent]:
+ """Keep deterministic explicit overrides ahead of hidden review."""
+ stripped = text.strip()
+ if not re.match(
+ r"^(?:remember:|invalidate:|forget:|please\s+remember\b|note\s+that\b)",
+ stripped,
+ re.IGNORECASE,
+ ):
+ return []
+ return self.extract_mutations(stripped)
+
+ def infer_query(self, text: str) -> CanonicalQueryIntent:
+ lowered = text.casefold()
+ labels = [
+ str(label) for valid, label in zip(self.store.valid.tolist(), self.store.cache.labels)
+ if valid and label and str(label).casefold() in lowered
+ ]
+ entity = max(labels, key=len) if labels else None
+ relation_id = None
+ if re.search(r"\b(owner|owns|owned|belongs)\b", lowered):
+ relation_id = 0
+ elif re.search(r"\b(location|located|where)\b", lowered):
+ relation_id = 1
+ elif re.search(r"\bstatus\b", lowered):
+ relation_id = 2
+ if entity is None:
+ query_patterns = (
+ r"\bwho\s+(?:owns|owned)\s+(?:the\s+)?(?P[^?.!]+)",
+ r"\b(?:where\s+is|location\s+of)\s+(?:the\s+)?(?P[^?.!]+)",
+ r"\bwhere\s+did\s+(?:i|we|my\s+character)\s+(?:leave|put|place|set)\s+"
+ r"(?:the\s+)?(?P[^?.!]+)",
+ r"\bstatus\s+of\s+(?:the\s+)?(?P[^?.!]+)",
+ )
+ for pattern in query_patterns:
+ match = re.search(pattern, text, re.IGNORECASE)
+ if match:
+ entity = self._clean_entity(match.group("entity"))
+ break
+ if entity is not None and not labels:
+ words = set(re.findall(r"[\w'-]+", entity.casefold()))
+ aliases = []
+ for valid, label in zip(self.store.valid.tolist(), self.store.cache.labels):
+ if not valid or not label:
+ continue
+ label_words = set(re.findall(r"[\w'-]+", str(label).casefold()))
+ if words and words <= label_words:
+ aliases.append(str(label))
+ if len(set(aliases)) == 1:
+ entity = aliases[0]
+ if entity is None:
+ return CanonicalQueryIntent(None, relation_id, None, "no known entity mentioned")
+ if relation_id is None:
+ relations = {
+ int(self.store.relation_id[index])
+ for index in self.store.valid.nonzero(as_tuple=False).flatten().tolist()
+ if self.store.cache.labels[index]
+ and str(self.store.cache.labels[index]).casefold() == entity.casefold()
+ }
+ if len(relations) == 1:
+ relation_id = next(iter(relations))
+ relation = None if relation_id is None else RELATION_NAMES[relation_id]
+ return CanonicalQueryIntent(entity, relation_id, relation, "explicit entity surface match")
+
+ def _matching_slots(self, entity: str, relation_id: int) -> list[int]:
+ return [
+ index for index in self.store.valid.nonzero(as_tuple=False).flatten().tolist()
+ if self.store.cache.labels[index]
+ and str(self.store.cache.labels[index]).casefold() == entity.casefold()
+ and int(self.store.relation_id[index]) == relation_id
+ ]
+
+ def apply(self, intents: Iterable[MutationIntent], recorder: SessionRecorder) -> None:
+ for intent in intents:
+ matches = self._matching_slots(intent.entity, intent.relation_id)
+ if intent.action == "invalidate":
+ if not matches:
+ recorder.event(
+ "P_IGNORE", source="p_cache", entity=intent.entity,
+ relation=RELATION_NAMES[intent.relation_id], reason="no active state to invalidate",
+ )
+ continue
+ for slot in matches:
+ before = self.entry(slot)
+ self.store.invalidate(slot)
+ self.surface_values.pop(slot, None)
+ self.translator_compatible.pop(slot, None)
+ recorder.event("P_INVALIDATE", source="p_cache", slot_id=slot, before=before)
+ continue
+ assert intent.value_id is not None and intent.value is not None
+ vector = self.vector(intent.entity, intent.relation_id, intent.value_id)
+ if matches:
+ slot = matches[0]
+ existing_surface = self.surface_values.get(slot, "").casefold()
+ if (
+ int(self.store.value_id[slot]) == intent.value_id
+ and existing_surface == intent.value.casefold()
+ ):
+ self.store.cache.keep(slot)
+ recorder.event(
+ "P_KEEP", source="p_cache", slot_id=slot,
+ entity=intent.entity, relation=RELATION_NAMES[intent.relation_id],
+ value=intent.value,
+ )
+ else:
+ before = self.entry(slot)
+ self.store.modify(
+ slot, vector, entity_id=self.entity_id(intent.entity),
+ relation_id=intent.relation_id, value_id=intent.value_id,
+ metadata_id=CANONICAL, confidence=1.0,
+ source=SlotSource.CORRECTION,
+ )
+ self.surface_values[slot] = intent.value
+ self.translator_compatible[slot] = intent.translator_compatible
+ recorder.event(
+ "P_MODIFY", source="p_cache", slot_id=slot, before=before,
+ after=self.entry(slot),
+ )
+ self.surface_values[slot] = intent.value
+ self.translator_compatible[slot] = intent.translator_compatible
+ continue
+ slot, operation = self.store.create(
+ vector, entity_id=self.entity_id(intent.entity),
+ relation_id=intent.relation_id, value_id=intent.value_id,
+ metadata_id=CANONICAL, slot_type=SlotType.FACT,
+ confidence=1.0, importance=0.7, freshness=Freshness.FRESH,
+ persistence=Persistence.SESSION, source=SlotSource.CONVERSATION,
+ label=intent.entity,
+ )
+ event = {
+ StateOperation.CREATE: "P_CREATE",
+ StateOperation.MERGE: "P_MERGE",
+ StateOperation.IGNORE: "P_IGNORE",
+ }[operation]
+ if slot >= 0:
+ self.surface_values[slot] = intent.value
+ self.translator_compatible[slot] = intent.translator_compatible
+ recorder.event(
+ event, source="p_cache", slot_id=None if slot < 0 else slot,
+ entity=intent.entity, relation=RELATION_NAMES[intent.relation_id],
+ value=intent.value,
+ translator_compatible=intent.translator_compatible,
+ )
+
+ def entry(self, index: int) -> dict[str, object]:
+ value_id = int(self.store.value_id[index])
+ return {
+ "slot_id": index,
+ "entity": self.store.cache.labels[index],
+ "entity_id": int(self.store.entity_id[index]),
+ "relation": RELATION_NAMES[int(self.store.relation_id[index])],
+ "relation_id": int(self.store.relation_id[index]),
+ "value": self.surface_values.get(
+ index,
+ CANONICAL_VALUE_LABELS[value_id]
+ if 0 <= value_id < len(CANONICAL_VALUE_LABELS) else None,
+ ),
+ "value_id": value_id,
+ "translator_compatible": self.translator_compatible.get(index, True),
+ "metadata_id": int(self.store.canonical_metadata_id[index]),
+ "confidence": float(self.store.cache.confidence[index]),
+ "importance": float(self.store.cache.importance[index]),
+ "freshness": Freshness(int(self.store.cache.freshness[index])).name.lower(),
+ "persistence": Persistence(int(self.store.cache.persistence[index])).name.lower(),
+ "source": SlotSource(int(self.store.cache.source[index])).name.lower(),
+ "last_updated": int(self.store.cache.last_updated[index]),
+ }
+
+ def snapshot(self) -> list[dict[str, object]]:
+ return [
+ self.entry(index)
+ for index in self.store.valid.nonzero(as_tuple=False).flatten().tolist()
+ ]
+
+ def translation_store(self, *, include_open_values: bool = False) -> CanonicalPStore:
+ compatible = [
+ index for index in self.store.valid.nonzero(as_tuple=False).flatten().tolist()
+ if include_open_values or self.translator_compatible.get(index, True)
+ ]
+ result = CanonicalPStore(CanonicalPConfig(
+ slots=max(1, len(compatible)), width=512, dtype=torch.float32,
+ device="cpu", merge_similarity=1.0,
+ ))
+ value_surfaces = {}
+ for index in compatible:
+ new_slot, _operation = result.create(
+ self.store.canonical_values[index],
+ entity_id=int(self.store.entity_id[index]),
+ relation_id=int(self.store.relation_id[index]),
+ value_id=int(self.store.value_id[index]),
+ metadata_id=int(self.store.canonical_metadata_id[index]),
+ slot_type=SlotType(int(self.store.cache.slot_type[index])),
+ confidence=float(self.store.cache.confidence[index]),
+ importance=float(self.store.cache.importance[index]),
+ freshness=Freshness(int(self.store.cache.freshness[index])),
+ persistence=Persistence(int(self.store.cache.persistence[index])),
+ source=SlotSource(int(self.store.cache.source[index])),
+ label=self.store.cache.labels[index],
+ )
+ if new_slot >= 0:
+ value_id = int(self.store.value_id[index])
+ value_surfaces[new_slot] = self.surface_values.get(
+ index,
+ CANONICAL_VALUE_LABELS[value_id]
+ if 0 <= value_id < len(CANONICAL_VALUE_LABELS) else None,
+ )
+ # Runtime-only canonical surface metadata. Canonical snapshots remain
+ # unchanged and never store model token identifiers or hidden vectors.
+ result._pcm_value_surfaces = value_surfaces
+ return result
+
+ def save_runtime_metadata(self, path: str | Path) -> None:
+ payload = {
+ "format": "pcm-interactive-p-metadata-v1",
+ "surface_values": {str(key): value for key, value in self.surface_values.items()},
+ "translator_compatible": {
+ str(key): value for key, value in self.translator_compatible.items()
+ },
+ }
+ Path(path).write_text(
+ json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=False) + "\n"
+ )
+
+ def load_runtime_metadata(self, path: str | Path) -> None:
+ source = Path(path)
+ if not source.is_file():
+ return
+ payload = json.loads(source.read_text())
+ if payload.get("format") != "pcm-interactive-p-metadata-v1":
+ raise ValueError("unsupported interactive P metadata")
+ self.surface_values.update({
+ int(key): str(value) for key, value in payload["surface_values"].items()
+ })
+ self.translator_compatible.update({
+ int(key): bool(value)
+ for key, value in payload["translator_compatible"].items()
+ })
+
+
+class PersonalityManager:
+ """One reusable `.ppkg` connection with explicit evidence extraction."""
+
+ def __init__(
+ self,
+ path: str | Path,
+ representation: FactorizedStateRepresentation,
+ *,
+ create: bool = True,
+ ) -> None:
+ self.path = Path(path)
+ if not self.path.exists():
+ if not create:
+ raise FileNotFoundError(self.path)
+ self.path.parent.mkdir(parents=True, exist_ok=True)
+ created = PersonalityPackage.create(
+ self.path, package_id=f"planner-personality-{safe_session_stamp(utc_now())}"
+ )
+ created.close()
+ self.package = PersonalityPackage(self.path, validate=True)
+ self.router = PersonalityRouter()
+ self.canonicalizer = FactorizedPersonalityCanonicalizer(
+ representation, value_labels=CANONICAL_VALUE_LABELS,
+ )
+ self.last_selection = None
+ self.mutations: list[dict[str, object]] = []
+
+ @staticmethod
+ def context(text: str) -> tuple[str, str]:
+ lowered = text.casefold()
+ if any(word in lowered for word in ("code", "debug", "error", "python", "cuda")):
+ return "technical", "debugging"
+ if any(word in lowered for word in ("story", "poem", "creative", "character")):
+ return "creative", "writing"
+ if any(word in lowered for word in ("roleplay", " rp ", "scene", "dialogue")):
+ return "roleplay", "roleplay"
+ return "chat", "general"
+
+ def extract_evidence(self, text: str, *, turn: int, timestamp: str) -> EvidenceRecord | None:
+ patterns = (
+ r"^i\s+(?:strongly\s+)?prefer\s+(?Pconcise|detailed|direct|structured|expressive)(?:\s+(?:responses|replies|answers))?[.!]?$",
+ r"^please\s+(?:always\s+)?(?:be|respond\s+in\s+a)\s+(?Pconcise|detailed|direct|structured|expressive)(?:\s+(?:style|way))?[.!]?$",
+ )
+ for pattern in patterns:
+ match = re.match(pattern, text.strip(), re.IGNORECASE)
+ if match:
+ interaction, domain = self.context(text)
+ return EvidenceRecord(
+ id=(
+ f"session-turn-{turn}-"
+ f"{hashlib.sha256((timestamp + text).encode()).hexdigest()[:16]}"
+ ),
+ entry_type=PersonalityType.INTERACTION_STYLE.value,
+ subject="user", relation="response_style",
+ value=match.group("value").casefold(), context=domain,
+ scope=interaction, confidence=0.95,
+ source_authority=EvidenceAuthority.EXPLICIT_USER.value,
+ timestamp=timestamp,
+ archive_reference=f"session://turn/{turn}/user",
+ )
+ return None
+
+ def ingest(self, record: EvidenceRecord, recorder: SessionRecorder) -> None:
+ change_count = len(self.package.changes())
+ decision = self.package.ingest(record)
+ payload = asdict(decision)
+ changes = self.package.changes()[change_count:]
+ self.mutations.append(payload)
+ recorder.event(
+ "PPKG_UPDATE", source="ppkg", evidence=asdict(record),
+ decision=payload, changes=changes,
+ )
+ if decision.promoted:
+ recorder.event("PPKG_PROMOTION", source="ppkg", **payload)
+ contradiction_changes = [
+ change for change in changes
+ if change["action"] in {"lower_confidence", "contradict", "supersede"}
+ ]
+ if contradiction_changes:
+ recorder.event(
+ "PPKG_CONTRADICTION", source="ppkg", decision=payload,
+ changes=contradiction_changes,
+ )
+
+ def query(
+ self,
+ text: str,
+ recorder: SessionRecorder,
+ *,
+ top_k: int = 4,
+ ) -> CanonicalPStore | None:
+ interaction, domain = self.context(text)
+ query = PersonalityQuery(
+ subject="user", interaction_type=interaction, domain=domain,
+ relation="response_style", timestamp=utc_now(),
+ )
+ recorder.event("PPKG_QUERY", source="ppkg", query=asdict(query), top_k=top_k)
+ selection = self.router.retrieve(self.package, query, top_k=top_k)
+ self.last_selection = selection
+ recorder.event(
+ "PPKG_CANDIDATES", source="ppkg",
+ candidate_count=selection.route.candidate_count,
+ accepted=selection.route.accepted,
+ entry_ids=list(selection.route.entry_ids), scores=list(selection.route.scores),
+ header_bytes_read=selection.route.header_bytes_read,
+ )
+ if not selection.entries:
+ return None
+ recorder.event(
+ "PPKG_LOAD", source="ppkg",
+ entries=[asdict(entry) for entry in selection.entries],
+ entry_bytes_read=selection.entry_bytes_read,
+ latency_seconds=selection.retrieval_latency_seconds,
+ translator_compatible_entries=[
+ entry.id for entry in selection.entries
+ if entry.value.casefold() in VALUE_LOOKUP
+ ],
+ )
+ compatible_entries = tuple(
+ entry for entry in selection.entries
+ if entry.value.casefold() in VALUE_LOOKUP
+ )
+ if not compatible_entries:
+ return None
+ store = CanonicalPStore(CanonicalPConfig(
+ slots=len(compatible_entries), width=512, dtype=torch.float32,
+ device="cpu", merge_similarity=1.0,
+ ))
+ for entry in compatible_entries:
+ vector, ids = self.canonicalizer.encode(entry)
+ store.create(
+ vector, entity_id=ids[0], relation_id=ids[1], value_id=ids[2],
+ metadata_id=ids[3], slot_type=SlotType.FACT,
+ confidence=entry.confidence, importance=entry.importance,
+ freshness=Freshness.FRESH, persistence=Persistence.DURABLE,
+ source=SlotSource.CONVERSATION, label=entry.subject,
+ )
+ return store
+
+ def visible_entries(
+ self, *, limit: int | None = None, offset: int = 0,
+ ) -> list[dict[str, object]]:
+ return [
+ asdict(entry)
+ for entry in self.package.entries(
+ status=PersonalityStatus.ACTIVE.value,
+ limit=limit,
+ offset=offset,
+ )
+ ]
+
+ def visible_entry_page(
+ self, *, limit: int = 100, offset: int = 0,
+ ) -> dict[str, object]:
+ """Return one bounded, deterministic debug page without loading the package."""
+ if limit <= 0 or limit > 200:
+ raise ValueError("personality debug limit must be between 1 and 200")
+ if offset < 0:
+ raise ValueError("personality debug offset cannot be negative")
+ total = self.package.entry_count(status=PersonalityStatus.ACTIVE.value)
+ entries = self.visible_entries(limit=limit, offset=offset)
+ returned = len(entries)
+ return {
+ "entries": entries,
+ "total_active": total,
+ "returned": returned,
+ "limit": limit,
+ "offset": offset,
+ "truncated": offset + returned < total,
+ }
+
+ def checkpoint(self) -> str:
+ return self.package.checkpoint(updated_at=utc_now())
+
+ def close(self) -> None:
+ self.package.close(checkpoint=True)
diff --git a/src/pcm/planner/memory_review.py b/src/pcm/planner/memory_review.py
new file mode 100644
index 0000000000000000000000000000000000000000..1129db56a72eb0a037ed35889a6eb96e0d3baabf
--- /dev/null
+++ b/src/pcm/planner/memory_review.py
@@ -0,0 +1,399 @@
+"""Fail-closed post-turn semantic review for the active canonical P-cache."""
+
+from __future__ import annotations
+
+from dataclasses import asdict, dataclass
+import json
+import re
+import time
+from typing import Callable
+
+from pcm.planner.interactive_session import (
+ CanonicalStateManager,
+ MutationIntent,
+ RELATION_NAMES,
+ SessionRecorder,
+)
+
+
+REVIEW_FORMAT = "planner-cache-memory-review-v1"
+REVIEW_CONFIDENCE_FLOOR = 0.80
+MAX_REVIEW_OPERATIONS = 8
+
+REVIEW_SOURCES = (
+ "explicit_user",
+ "user_correction",
+ "rp_action",
+ "tool_verified",
+ "assistant_inference",
+ "assistant_unsupported",
+)
+AUTHORITATIVE_REVIEW_SOURCES = {
+ "explicit_user", "user_correction", "rp_action",
+}
+
+MEMORY_REVIEW_SCHEMA: dict[str, object] = {
+ "type": "object",
+ "additionalProperties": False,
+ "required": ["operations"],
+ "properties": {
+ "operations": {
+ "type": "array",
+ "maxItems": MAX_REVIEW_OPERATIONS,
+ "items": {
+ "type": "object",
+ "additionalProperties": False,
+ "required": [
+ "op", "entity", "relation", "value", "confidence", "source",
+ ],
+ "properties": {
+ "op": {"enum": [
+ "CREATE", "MODIFY", "KEEP", "MERGE",
+ "INVALIDATE", "IGNORE",
+ ]},
+ "entity": {"type": ["string", "null"], "maxLength": 160},
+ "relation": {
+ "type": ["string", "null"],
+ "enum": ["owner", "location", "status", None],
+ },
+ "value": {"type": ["string", "null"], "maxLength": 160},
+ "confidence": {"type": "number", "minimum": 0, "maximum": 1},
+ "source": {"enum": list(REVIEW_SOURCES)},
+ },
+ },
+ },
+ },
+}
+
+
+@dataclass(frozen=True)
+class ReviewedOperation:
+ op: str
+ entity: str | None
+ relation: str | None
+ value: str | None
+ confidence: float
+ source: str
+
+
+@dataclass(frozen=True)
+class ValidatedReview:
+ proposed: tuple[ReviewedOperation, ...]
+ accepted: tuple[ReviewedOperation, ...]
+ intents: tuple[MutationIntent, ...]
+ rejected: tuple[dict[str, object], ...]
+
+
+def review_prompt(
+ *,
+ user_message: str,
+ assistant_response: str,
+ recent_context: list[tuple[str, str]],
+ relevant_state: list[dict[str, object]],
+) -> str:
+ rules = """You are the hidden Planner Cache memory reviewer.
+Return one JSON object matching the supplied schema and no prose.
+P-cache stores current useful semantic state, never transcript summaries.
+Allowed relations are owner, location, and status.
+CREATE a new current fact. MODIFY a changed current value. KEEP an unchanged
+current fact. MERGE only equivalent duplicate state. INVALIDATE a fact the user
+explicitly says is no longer valid. IGNORE transient chat and unsupported claims.
+Treat explicit user corrections as highest authority. RP actions directly written
+by the user are authoritative current events. Never promote assistant inventions,
+inferences, suggestions, jokes, or unsupported generated claims. Use source
+assistant_inference or assistant_unsupported for such candidates so validation can
+reject them. Resolve pronouns only when recent context or current P makes the
+referent unambiguous. Prefer IGNORE when uncertain. Historical values must not
+remain active beside their corrected current value.
+The source value must be exactly one of: explicit_user, user_correction,
+rp_action, tool_verified, assistant_inference, assistant_unsupported."""
+ payload = {
+ "format": REVIEW_FORMAT,
+ "newest_user_message": user_message[-2000:],
+ "newest_assistant_response_untrusted": assistant_response[-1000:],
+ "limited_recent_context": [
+ {"user": user[-600:], "assistant": assistant[-600:]}
+ for user, assistant in recent_context[-2:]
+ ],
+ "relevant_current_p": relevant_state,
+ "required_output": {
+ "operations": [{
+ "op": "CREATE|MODIFY|KEEP|MERGE|INVALIDATE|IGNORE",
+ "entity": "string|null",
+ "relation": "owner|location|status|null",
+ "value": "string|null",
+ "confidence": "number 0..1",
+ "source": "one allowed authority label from the rules",
+ }],
+ },
+ }
+ return rules + "\n\nINPUT:\n" + json.dumps(
+ payload, ensure_ascii=False, sort_keys=True,
+ )
+
+
+def parse_review(raw: str) -> tuple[ReviewedOperation, ...]:
+ if raw != raw.strip():
+ raw = raw.strip()
+ value = json.loads(raw)
+ if not isinstance(value, dict) or set(value) != {"operations"}:
+ raise ValueError("review must contain only operations")
+ operations = value["operations"]
+ if not isinstance(operations, list) or len(operations) > MAX_REVIEW_OPERATIONS:
+ raise ValueError("review operations must be a bounded array")
+ parsed = []
+ required = {"op", "entity", "relation", "value", "confidence", "source"}
+ for item in operations:
+ if not isinstance(item, dict) or set(item) != required:
+ raise ValueError("review operation has an invalid shape")
+ op = str(item["op"]).upper()
+ if op not in {"CREATE", "MODIFY", "KEEP", "MERGE", "INVALIDATE", "IGNORE"}:
+ raise ValueError("unsupported review operation")
+ source = str(item["source"])
+ if source not in REVIEW_SOURCES:
+ raise ValueError("unsupported review source")
+ confidence = float(item["confidence"])
+ if not 0 <= confidence <= 1:
+ raise ValueError("review confidence must be between zero and one")
+ entity = item["entity"]
+ relation = item["relation"]
+ candidate_value = item["value"]
+ if entity is not None and not isinstance(entity, str):
+ raise ValueError("entity must be a string or null")
+ if relation is not None and relation not in RELATION_NAMES:
+ raise ValueError("unsupported canonical relation")
+ if candidate_value is not None and not isinstance(candidate_value, str):
+ raise ValueError("value must be a string or null")
+ parsed.append(ReviewedOperation(
+ op=op,
+ entity=None if entity is None else entity.strip()[:160],
+ relation=relation,
+ value=None if candidate_value is None else candidate_value.strip()[:160],
+ confidence=confidence,
+ source=source,
+ ))
+ return tuple(parsed)
+
+
+def _words(text: str) -> set[str]:
+ return set(re.findall(r"[\w'-]+", text.casefold()))
+
+
+def relevant_state_for_review(
+ state: CanonicalStateManager,
+ user_message: str,
+ recent_context: list[tuple[str, str]],
+ *,
+ limit: int = 12,
+) -> list[dict[str, object]]:
+ context = " ".join(
+ [user_message]
+ + [part for turn in recent_context[-2:] for part in turn]
+ )
+ context_words = _words(context)
+ entries = state.snapshot()
+ ranked = sorted(
+ entries,
+ key=lambda entry: (
+ bool(_words(str(entry.get("entity", ""))) & context_words),
+ int(entry.get("slot_id", -1)),
+ ),
+ reverse=True,
+ )
+ relevant = [
+ entry for entry in ranked
+ if _words(str(entry.get("entity", ""))) & context_words
+ ]
+ if not relevant:
+ relevant = ranked[:4]
+ return [
+ {
+ key: entry.get(key)
+ for key in ("slot_id", "entity", "relation", "value", "confidence", "source")
+ }
+ for entry in relevant[:limit]
+ ]
+
+
+def validate_review(
+ operations: tuple[ReviewedOperation, ...],
+ state: CanonicalStateManager,
+) -> ValidatedReview:
+ accepted = []
+ intents = []
+ rejected = []
+ for operation in operations:
+ reason = None
+ if operation.op == "IGNORE":
+ continue
+ if operation.source not in AUTHORITATIVE_REVIEW_SOURCES:
+ reason = "assistant or unsupported evidence cannot mutate P"
+ elif operation.confidence < REVIEW_CONFIDENCE_FLOOR:
+ reason = "confidence below conservative review threshold"
+ elif not operation.entity or not operation.relation:
+ reason = "entity and canonical relation are required"
+ elif operation.op != "INVALIDATE" and not operation.value:
+ reason = "value is required for current-state mutation"
+ if reason is not None:
+ rejected.append({"operation": asdict(operation), "reason": reason})
+ continue
+ relation_id = RELATION_NAMES.index(operation.relation)
+ matches = state._matching_slots(operation.entity, relation_id)
+ if operation.op == "INVALIDATE":
+ if not matches:
+ rejected.append({
+ "operation": asdict(operation),
+ "reason": "no matching active state to invalidate",
+ })
+ continue
+ intent = MutationIntent("invalidate", operation.entity, relation_id)
+ else:
+ value_id, surface, compatible = state._canonical_value(operation.value or "")
+ exact = any(
+ int(state.store.value_id[slot]) == value_id
+ and state.surface_values.get(slot, "").casefold() == surface.casefold()
+ for slot in matches
+ )
+ consistency_reason = None
+ if operation.op == "CREATE" and matches:
+ consistency_reason = "CREATE cannot replace existing current state"
+ elif operation.op == "MODIFY" and (not matches or exact):
+ consistency_reason = "MODIFY requires a changed existing current value"
+ elif operation.op in {"KEEP", "MERGE"} and not exact:
+ consistency_reason = (
+ f"{operation.op} requires equivalent existing current state"
+ )
+ if consistency_reason is not None:
+ rejected.append({
+ "operation": asdict(operation), "reason": consistency_reason,
+ })
+ continue
+ intent = MutationIntent(
+ "upsert", operation.entity, relation_id,
+ value_id, surface, compatible,
+ )
+ accepted.append(operation)
+ intents.append(intent)
+ return ValidatedReview(
+ proposed=operations,
+ accepted=tuple(accepted),
+ intents=tuple(intents),
+ rejected=tuple(rejected),
+ )
+
+
+class PostTurnMemoryReviewer:
+ """Run one serialized same-model review and apply only validated user state."""
+
+ def __init__(self, generate: Callable[[str, dict[str, object]], str]) -> None:
+ self.generate = generate
+
+ def run(
+ self,
+ *,
+ user_message: str,
+ assistant_response: str,
+ recent_context: list[tuple[str, str]],
+ state: CanonicalStateManager,
+ recorder: SessionRecorder,
+ ) -> ValidatedReview | None:
+ relevant = relevant_state_for_review(state, user_message, recent_context)
+ prompt = review_prompt(
+ user_message=user_message,
+ assistant_response=assistant_response,
+ recent_context=recent_context,
+ relevant_state=relevant,
+ )
+ started = time.perf_counter()
+ recorder.event(
+ "MEMORY_REVIEW_START", source="memory_review",
+ source_turn=recorder.turn, relevant_state=relevant,
+ recent_context_turns=min(2, len(recent_context)),
+ )
+ before = state.snapshot()
+ raw = ""
+ try:
+ raw = self.generate(prompt, MEMORY_REVIEW_SCHEMA)
+ operations = parse_review(raw)
+ validated = validate_review(operations, state)
+ except Exception as error:
+ recorder.event(
+ "MEMORY_REVIEW_REJECTED", source="memory_review",
+ source_turn=recorder.turn, validation="malformed_or_runtime_error",
+ error_type=type(error).__name__, message=str(error),
+ raw_output=raw[:4000],
+ latency_seconds=time.perf_counter() - started,
+ )
+ return None
+ recorder.event(
+ "MEMORY_REVIEW_RESULT", source="memory_review",
+ source_turn=recorder.turn,
+ proposed_operations=[asdict(item) for item in validated.proposed],
+ accepted_operations=[asdict(item) for item in validated.accepted],
+ rejected_operations=list(validated.rejected),
+ validation="accepted" if validated.intents else "no_mutation",
+ latency_seconds=time.perf_counter() - started,
+ )
+ if validated.rejected:
+ recorder.event(
+ "MEMORY_REVIEW_REJECTED", source="memory_review",
+ source_turn=recorder.turn,
+ validation="operation_rejection",
+ rejected_operations=list(validated.rejected),
+ )
+ if validated.intents:
+ cache = state.store.cache
+ tensor_owners = (
+ (cache, (
+ "values", "valid", "slot_type", "confidence", "importance",
+ "freshness", "persistence", "last_updated", "source",
+ )),
+ (state.store, (
+ "entity_id", "relation_id", "value_id", "canonical_metadata_id",
+ )),
+ )
+ tensors = {
+ (id(owner), name): getattr(owner, name).clone()
+ for owner, names in tensor_owners for name in names
+ }
+ labels = list(cache.labels)
+ clock = cache._clock
+ surfaces = dict(state.surface_values)
+ compatibility = dict(state.translator_compatible)
+ buffered_events: list[tuple[str, dict[str, object]]] = []
+
+ class BufferedRecorder:
+ @staticmethod
+ def event(event: str, **fields: object) -> None:
+ buffered_events.append((event, fields))
+
+ try:
+ state.apply(validated.intents, BufferedRecorder())
+ except Exception as error:
+ for owner, names in tensor_owners:
+ for name in names:
+ getattr(owner, name).copy_(tensors[(id(owner), name)])
+ cache.labels = labels
+ cache._clock = clock
+ state.surface_values = surfaces
+ state.translator_compatible = compatibility
+ recorder.event(
+ "MEMORY_REVIEW_REJECTED", source="memory_review",
+ source_turn=recorder.turn,
+ validation="atomic_apply_failed",
+ error_type=type(error).__name__, message=str(error),
+ latency_seconds=time.perf_counter() - started,
+ )
+ return None
+ for event, fields in buffered_events:
+ recorder.event(event, **fields)
+ after = state.snapshot()
+ recorder.event(
+ "MEMORY_REVIEW_APPLIED", source="memory_review",
+ source_turn=recorder.turn,
+ applied_mutations=[asdict(item) for item in validated.intents],
+ before=before, after=after,
+ changed=before != after,
+ latency_seconds=time.perf_counter() - started,
+ )
+ return validated
diff --git a/src/pcm/planner/personality.py b/src/pcm/planner/personality.py
new file mode 100644
index 0000000000000000000000000000000000000000..6f40e29e85cf4665aea8f620aad1aee7c6fa69bc
--- /dev/null
+++ b/src/pcm/planner/personality.py
@@ -0,0 +1,1180 @@
+"""Disk-resident, model-independent durable personality memory.
+
+The package is deliberately mechanical: evidence is accumulated on CPU/disk,
+promotion is deterministic, and only selected canonical entries are activated
+through the existing P translation boundary.
+"""
+
+from __future__ import annotations
+
+from dataclasses import asdict, dataclass, replace
+from datetime import datetime, timezone
+from enum import Enum
+import hashlib
+import json
+import math
+import os
+from pathlib import Path
+import sqlite3
+from typing import Iterable, Sequence
+
+import torch
+from torch import Tensor
+import torch.nn.functional as F
+
+from pcm.planner.canonical import CanonicalPConfig, CanonicalPStore
+from pcm.planner.cache import Freshness, Persistence, SlotSource, SlotType
+from pcm.planner.representation import CANONICAL, FactorizedStateRepresentation
+from pcm.planner.split_translator import ByteEntityEncoder
+from pcm.planner.canonical import CANONICAL_P_PROTOCOL
+
+
+PPKG_FORMAT = "pcm-personality-package-v1"
+PPKG_PROTOCOL = "pcm-canonical-personality-v1"
+
+
+class PersonalityType(str, Enum):
+ TRAIT = "trait"
+ PREFERENCE = "preference"
+ RELATIONSHIP_PATTERN = "relationship_pattern"
+ BEHAVIORAL_PATTERN = "behavioral_pattern"
+ INTERACTION_STYLE = "interaction_style"
+ TERMINOLOGY = "terminology"
+ HABIT = "habit"
+ RESPONSE_TENDENCY = "response_tendency"
+ CONTEXTUAL_TENDENCY = "contextual_tendency"
+
+
+class EvidenceAuthority(str, Enum):
+ EXPLICIT_USER = "explicit_user_correction_or_statement"
+ EXTERNALLY_VERIFIED = "externally_verified_observation"
+ REPEATED_OBSERVED = "repeated_observed_interaction_behavior"
+ SINGLE_OBSERVED = "single_observed_behavior"
+ MODEL_INFERENCE = "model_inference"
+ MODEL_UNSUPPORTED = "model_generated_unsupported_claim"
+
+ @property
+ def weight(self) -> float:
+ return {
+ EvidenceAuthority.EXPLICIT_USER: 1.0,
+ EvidenceAuthority.EXTERNALLY_VERIFIED: 0.9,
+ EvidenceAuthority.REPEATED_OBSERVED: 0.7,
+ EvidenceAuthority.SINGLE_OBSERVED: 0.45,
+ EvidenceAuthority.MODEL_INFERENCE: 0.15,
+ EvidenceAuthority.MODEL_UNSUPPORTED: 0.0,
+ }[self]
+
+
+class PersonalityStatus(str, Enum):
+ ACTIVE = "active"
+ CONTRADICTED = "contradicted"
+ SUPERSEDED = "superseded"
+ INVALIDATED = "invalidated"
+
+
+@dataclass(frozen=True)
+class EvidenceRecord:
+ id: str
+ entry_type: str
+ subject: str
+ relation: str
+ value: str
+ context: str
+ scope: str
+ confidence: float
+ source_authority: str
+ timestamp: str
+ archive_reference: str
+ polarity: int = 1
+ relationship: str | None = None
+ connected_evidence_ids: tuple[str, ...] = ()
+ note: str | None = None
+
+ def __post_init__(self) -> None:
+ if not self.id or not self.subject or not self.relation or not self.value:
+ raise ValueError("evidence id, subject, relation, and value are required")
+ if not 0 <= self.confidence <= 1:
+ raise ValueError("evidence confidence must be in [0, 1]")
+ if self.polarity not in (-1, 1):
+ raise ValueError("evidence polarity must be -1 or 1")
+ EvidenceAuthority(self.source_authority)
+ if not self.archive_reference:
+ raise ValueError("evidence must reference an archive/source event")
+
+
+@dataclass(frozen=True)
+class PersonalityEntry:
+ id: str
+ entry_type: str
+ subject: str
+ relation: str
+ value: str
+ scope: str
+ relationship: str | None
+ strength: float
+ confidence: float
+ importance: float
+ evidence_count: int
+ context_diversity: int
+ last_reinforced: str
+ created_at: str
+ updated_at: str
+ source_authority: str
+ supporting_evidence_ids: tuple[str, ...]
+ contradicting_evidence_ids: tuple[str, ...]
+ status: str = PersonalityStatus.ACTIVE.value
+ extension: dict[str, object] | None = None
+
+
+@dataclass(frozen=True)
+class PromotionPolicy:
+ """Public coefficients for the conservative v1 promotion score."""
+
+ promotion_threshold: float = 1.25
+ context_diversity_coefficient: float = 0.8
+ connectivity_coefficient: float = 0.25
+ contradiction_weight: float = 0.75
+ confidence_prior: float = 0.5
+ explicit_override_confidence: float = 0.85
+
+ def score(self, evidence: Sequence[EvidenceRecord]) -> dict[str, float]:
+ supporting = [record for record in evidence if record.polarity > 0]
+ weighted_support = sum(
+ record.confidence * EvidenceAuthority(record.source_authority).weight
+ for record in supporting
+ )
+ diversity = len({record.context for record in supporting})
+ linked = len({
+ linked_id
+ for record in supporting
+ for linked_id in record.connected_evidence_ids
+ })
+ diversity_factor = 1 + self.context_diversity_coefficient * math.log1p(
+ max(0, diversity - 1)
+ )
+ connectivity_factor = 1 + self.connectivity_coefficient * math.log1p(linked)
+ promotion_score = weighted_support * diversity_factor * connectivity_factor
+ return {
+ "weighted_support": weighted_support,
+ "context_diversity": float(diversity),
+ "connected_evidence": float(linked),
+ "diversity_factor": diversity_factor,
+ "connectivity_factor": connectivity_factor,
+ "promotion_score": promotion_score,
+ }
+
+
+@dataclass(frozen=True)
+class PromotionDecision:
+ evidence_id: str
+ promoted: bool
+ entry_id: str | None
+ action: str
+ promotion_score: float
+ threshold: float
+ reason: str
+
+
+@dataclass(frozen=True)
+class PersonalityQuery:
+ subject: str
+ interaction_type: str
+ domain: str
+ relationship: str | None = None
+ relation: str | None = None
+ timestamp: str | None = None
+
+
+@dataclass(frozen=True)
+class PersonalityRoute:
+ entry_ids: tuple[str, ...]
+ scores: tuple[float, ...]
+ candidate_count: int
+ header_bytes_read: int
+ accepted: int
+
+
+@dataclass(frozen=True)
+class PersonalitySelection:
+ entries: tuple[PersonalityEntry, ...]
+ route: PersonalityRoute
+ entry_bytes_read: int
+ retrieval_latency_seconds: float
+
+ @property
+ def logical_bytes_read(self) -> int:
+ return self.route.header_bytes_read + self.entry_bytes_read
+
+
+def _canonical_json(value: object) -> str:
+ return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
+
+
+def _utc_now() -> str:
+ return datetime.now(timezone.utc).isoformat(timespec="microseconds")
+
+
+def _parse_time(value: str) -> datetime:
+ parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
+ return parsed if parsed.tzinfo else parsed.replace(tzinfo=timezone.utc)
+
+
+def _stable_id(prefix: str, *fields: object) -> str:
+ digest = hashlib.sha256(_canonical_json(fields).encode()).hexdigest()[:24]
+ return f"{prefix}_{digest}"
+
+
+def _entry_to_row(entry: PersonalityEntry) -> tuple[object, ...]:
+ return (
+ entry.id, entry.entry_type, entry.subject, entry.relation, entry.value,
+ entry.scope, entry.relationship, entry.strength, entry.confidence,
+ entry.importance, entry.evidence_count, entry.context_diversity,
+ entry.last_reinforced, entry.created_at, entry.updated_at,
+ entry.source_authority, _canonical_json(entry.supporting_evidence_ids),
+ _canonical_json(entry.contradicting_evidence_ids), entry.status,
+ _canonical_json(entry.extension or {}),
+ )
+
+
+def _row_to_entry(row: Sequence[object]) -> PersonalityEntry:
+ return PersonalityEntry(
+ id=str(row[0]), entry_type=str(row[1]), subject=str(row[2]),
+ relation=str(row[3]), value=str(row[4]), scope=str(row[5]),
+ relationship=None if row[6] is None else str(row[6]),
+ strength=float(row[7]), confidence=float(row[8]), importance=float(row[9]),
+ evidence_count=int(row[10]), context_diversity=int(row[11]),
+ last_reinforced=str(row[12]), created_at=str(row[13]), updated_at=str(row[14]),
+ source_authority=str(row[15]),
+ supporting_evidence_ids=tuple(json.loads(str(row[16]))),
+ contradicting_evidence_ids=tuple(json.loads(str(row[17]))),
+ status=str(row[18]), extension=json.loads(str(row[19])),
+ )
+
+
+ENTRY_COLUMNS = (
+ "id,type,subject,relation,value,scope,relationship,strength,confidence,importance,"
+ "evidence_count,context_diversity,last_reinforced,created_at,updated_at,"
+ "source_authority,supporting_ids,contradicting_ids,status,extension_json"
+)
+
+
+class PersonalityPackage:
+ """A SQLite-backed `.ppkg`; opening it allocates no CUDA tensors."""
+
+ def __init__(self, path: str | Path, *, validate: bool = True) -> None:
+ self.path = Path(path)
+ if not self.path.is_file():
+ raise FileNotFoundError(self.path)
+ # Web sessions create the package on the launcher thread and execute
+ # serialized turns on the HTTP worker thread. SQLite's transactional
+ # behavior is unchanged. Cross-thread access is enabled only so the
+ # gateway's single session lock can own that serialization boundary.
+ self._connection = sqlite3.connect(self.path, check_same_thread=False)
+ self._connection.row_factory = sqlite3.Row
+ self._connection.execute("PRAGMA query_only = ON")
+ self._closed = False
+ self._validate_header()
+ self._dirty = self.metadata().get("integrity_state", "clean") != "clean"
+ if validate:
+ self.validate_checksum()
+
+ @classmethod
+ def create(
+ cls,
+ path: str | Path,
+ *,
+ package_id: str,
+ created_at: str | None = None,
+ metadata: dict[str, object] | None = None,
+ overwrite: bool = False,
+ ) -> "PersonalityPackage":
+ path = Path(path)
+ if path.exists() and not overwrite:
+ raise FileExistsError(path)
+ if path.exists():
+ path.unlink()
+ path.parent.mkdir(parents=True, exist_ok=True)
+ connection = sqlite3.connect(path)
+ connection.executescript(
+ """
+ PRAGMA page_size = 4096;
+ PRAGMA journal_mode = DELETE;
+ PRAGMA synchronous = FULL;
+ CREATE TABLE metadata (key TEXT PRIMARY KEY, value TEXT NOT NULL);
+ CREATE TABLE entries (
+ id TEXT PRIMARY KEY, type TEXT NOT NULL, subject TEXT NOT NULL,
+ relation TEXT NOT NULL, value TEXT NOT NULL, scope TEXT NOT NULL,
+ relationship TEXT, strength REAL NOT NULL, confidence REAL NOT NULL,
+ importance REAL NOT NULL, evidence_count INTEGER NOT NULL,
+ context_diversity INTEGER NOT NULL, last_reinforced TEXT NOT NULL,
+ created_at TEXT NOT NULL, updated_at TEXT NOT NULL,
+ source_authority TEXT NOT NULL, supporting_ids TEXT NOT NULL,
+ contradicting_ids TEXT NOT NULL, status TEXT NOT NULL,
+ extension_json TEXT NOT NULL
+ );
+ CREATE INDEX entry_lookup ON entries(status, subject, relation, scope, relationship);
+ CREATE INDEX entry_subject_route ON entries(
+ status,subject,importance DESC,confidence DESC,id
+ );
+ CREATE INDEX entry_scope_route ON entries(
+ status,scope,importance DESC,confidence DESC,id
+ );
+ CREATE INDEX entry_relationship_route ON entries(
+ status,relationship,importance DESC,confidence DESC,id
+ );
+ CREATE TABLE evidence (
+ id TEXT PRIMARY KEY, entry_type TEXT NOT NULL, subject TEXT NOT NULL,
+ relation TEXT NOT NULL, value TEXT NOT NULL, context TEXT NOT NULL,
+ scope TEXT NOT NULL, confidence REAL NOT NULL,
+ source_authority TEXT NOT NULL, timestamp TEXT NOT NULL,
+ archive_reference TEXT NOT NULL, polarity INTEGER NOT NULL,
+ relationship TEXT, connected_ids TEXT NOT NULL, note TEXT
+ );
+ CREATE INDEX evidence_candidate ON evidence(entry_type, subject, relation, scope, relationship, value);
+ CREATE TABLE changes (
+ sequence INTEGER PRIMARY KEY AUTOINCREMENT, transaction_id TEXT NOT NULL,
+ timestamp TEXT NOT NULL, action TEXT NOT NULL, entry_id TEXT NOT NULL,
+ before_json TEXT, after_json TEXT
+ );
+ """
+ )
+ stamp = created_at or _utc_now()
+ values = {
+ "format": PPKG_FORMAT,
+ "protocol": PPKG_PROTOCOL,
+ "canonical_p_protocol": CANONICAL_P_PROTOCOL,
+ "package_id": package_id,
+ "created_at": stamp,
+ "updated_at": stamp,
+ "schema_version": "1",
+ "extensions": _canonical_json(metadata or {}),
+ "integrity_state": "clean",
+ "content_sha256": "pending",
+ }
+ connection.executemany(
+ "INSERT INTO metadata(key,value) VALUES (?,?)", sorted(values.items())
+ )
+ connection.commit()
+ connection.close()
+ package = cls(path, validate=False)
+ package._dirty = True
+ package.checkpoint(updated_at=stamp)
+ return package
+
+ def close(self, *, checkpoint: bool = True) -> None:
+ if self._closed:
+ return
+ if checkpoint and self._dirty:
+ self.checkpoint()
+ self._connection.close()
+ self._closed = True
+
+ def __enter__(self) -> "PersonalityPackage":
+ return self
+
+ def __exit__(self, *_exc) -> None:
+ self.close()
+
+ @property
+ def package_id(self) -> str:
+ return self.metadata()["package_id"]
+
+ @property
+ def size_bytes(self) -> int:
+ return self.path.stat().st_size
+
+ def metadata(self) -> dict[str, str]:
+ return {
+ str(row[0]): str(row[1])
+ for row in self._connection.execute("SELECT key,value FROM metadata")
+ }
+
+ def _validate_header(self) -> None:
+ try:
+ metadata = self.metadata()
+ except sqlite3.DatabaseError as error:
+ raise ValueError("corrupt personality package") from error
+ if metadata.get("format") != PPKG_FORMAT:
+ raise ValueError("unsupported personality package format")
+ if metadata.get("protocol") != PPKG_PROTOCOL:
+ raise ValueError("incompatible personality protocol")
+ if metadata.get("canonical_p_protocol") != CANONICAL_P_PROTOCOL:
+ raise ValueError("incompatible canonical P protocol")
+ if metadata.get("schema_version") != "1":
+ raise ValueError("unsupported personality package schema")
+
+ def _semantic_checksum(self) -> str:
+ digest = hashlib.sha256()
+ metadata = [
+ tuple(row)
+ for row in self._connection.execute(
+ "SELECT key,value FROM metadata WHERE key != 'content_sha256' ORDER BY key"
+ )
+ ]
+ digest.update(_canonical_json(metadata).encode())
+ for table, order in (
+ ("entries", "id"), ("evidence", "id"), ("changes", "sequence")
+ ):
+ for row in self._connection.execute(f"SELECT * FROM {table} ORDER BY {order}"):
+ digest.update(_canonical_json(tuple(row)).encode())
+ return digest.hexdigest()
+
+ def validate_checksum(self) -> None:
+ if self._dirty or self.metadata().get("integrity_state", "clean") != "clean":
+ raise ValueError("personality package has uncheckpointed changes")
+ expected = self.metadata().get("content_sha256")
+ try:
+ actual = self._semantic_checksum()
+ except sqlite3.DatabaseError as error:
+ raise ValueError("corrupt personality package") from error
+ if not expected or expected != actual:
+ raise ValueError("personality package checksum does not match")
+
+ def verify(self) -> None:
+ """Explicit full-package integrity boundary."""
+ self.validate_checksum()
+
+ def _writable(self) -> None:
+ self._connection.execute("PRAGMA query_only = OFF")
+
+ def _readonly(self) -> None:
+ self._connection.execute("PRAGMA query_only = ON")
+
+ def _mark_dirty(self, updated_at: str) -> None:
+ self._connection.execute(
+ "UPDATE metadata SET value=? WHERE key='updated_at'", (updated_at,)
+ )
+ self._connection.execute(
+ "INSERT OR REPLACE INTO metadata(key,value) VALUES ('integrity_state','dirty')"
+ )
+ self._dirty = True
+
+ def checkpoint(self, *, updated_at: str | None = None) -> str:
+ """Atomically seal all committed mutations with one semantic checksum."""
+ if self._closed:
+ raise RuntimeError("personality package is closed")
+ if not self._dirty:
+ return self.metadata()["content_sha256"]
+ self._writable()
+ try:
+ self._connection.execute("BEGIN IMMEDIATE")
+ self._connection.execute(
+ "UPDATE metadata SET value=? WHERE key='updated_at'",
+ (updated_at or self.metadata()["updated_at"],),
+ )
+ self._connection.execute(
+ "INSERT OR REPLACE INTO metadata(key,value) VALUES ('integrity_state','clean')"
+ )
+ self._connection.execute(
+ "UPDATE metadata SET value='pending' WHERE key='content_sha256'"
+ )
+ checksum = self._semantic_checksum()
+ self._connection.execute(
+ "UPDATE metadata SET value=? WHERE key='content_sha256'", (checksum,)
+ )
+ self._connection.commit()
+ except Exception:
+ self._connection.rollback()
+ self._readonly()
+ raise
+ self._dirty = False
+ self._readonly()
+ return checksum
+
+ def export(self, path: str | Path) -> Path:
+ """Checkpoint, snapshot with SQLite backup, and verify the snapshot."""
+ self.checkpoint()
+ destination = Path(path)
+ if destination.exists():
+ raise FileExistsError(destination)
+ target = sqlite3.connect(destination)
+ try:
+ self._connection.backup(target)
+ finally:
+ target.close()
+ with PersonalityPackage(destination, validate=True):
+ pass
+ return destination
+
+ def _evidence_rows(self, record: EvidenceRecord) -> None:
+ self._connection.execute(
+ """INSERT INTO evidence VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)""",
+ (
+ record.id, record.entry_type, record.subject, record.relation,
+ record.value, record.context, record.scope, record.confidence,
+ record.source_authority, record.timestamp, record.archive_reference,
+ record.polarity, record.relationship,
+ _canonical_json(record.connected_evidence_ids), record.note,
+ ),
+ )
+
+ def evidence(self, evidence_id: str) -> EvidenceRecord:
+ row = self._connection.execute(
+ "SELECT * FROM evidence WHERE id=?", (evidence_id,)
+ ).fetchone()
+ if row is None:
+ raise KeyError(evidence_id)
+ return EvidenceRecord(
+ id=row[0], entry_type=row[1], subject=row[2], relation=row[3], value=row[4],
+ context=row[5], scope=row[6], confidence=row[7], source_authority=row[8],
+ timestamp=row[9], archive_reference=row[10], polarity=row[11],
+ relationship=row[12], connected_evidence_ids=tuple(json.loads(row[13])),
+ note=row[14],
+ )
+
+ def _candidate_records(self, record: EvidenceRecord) -> list[EvidenceRecord]:
+ rows = self._connection.execute(
+ """SELECT id FROM evidence WHERE entry_type=? AND subject=? AND relation=?
+ AND scope=? AND relationship IS ? ORDER BY id""",
+ (record.entry_type, record.subject, record.relation, record.scope, record.relationship),
+ )
+ return [self.evidence(str(row[0])) for row in rows]
+
+ def entry_count(self, *, status: str | None = None) -> int:
+ if status is None:
+ row = self._connection.execute("SELECT COUNT(*) FROM entries").fetchone()
+ else:
+ row = self._connection.execute(
+ "SELECT COUNT(*) FROM entries WHERE status=?", (status,)
+ ).fetchone()
+ assert row is not None
+ return int(row[0])
+
+ def entries(
+ self,
+ *,
+ status: str | None = None,
+ limit: int | None = None,
+ offset: int = 0,
+ ) -> list[PersonalityEntry]:
+ if limit is not None and limit <= 0:
+ raise ValueError("entry limit must be positive")
+ if offset < 0:
+ raise ValueError("entry offset cannot be negative")
+ where = "" if status is None else " WHERE status=?"
+ parameters: list[object] = [] if status is None else [status]
+ pagination = ""
+ if limit is not None:
+ pagination = " LIMIT ? OFFSET ?"
+ parameters.extend((limit, offset))
+ elif offset:
+ pagination = " LIMIT -1 OFFSET ?"
+ parameters.append(offset)
+ rows = self._connection.execute(
+ f"SELECT {ENTRY_COLUMNS} FROM entries{where} ORDER BY id{pagination}",
+ parameters,
+ )
+ return [_row_to_entry(tuple(row)) for row in rows]
+
+ def entry(self, entry_id: str) -> PersonalityEntry:
+ row = self._connection.execute(
+ f"SELECT {ENTRY_COLUMNS} FROM entries WHERE id=?", (entry_id,)
+ ).fetchone()
+ if row is None:
+ raise KeyError(entry_id)
+ return _row_to_entry(tuple(row))
+
+ def routing_headers(
+ self,
+ *,
+ subject: str,
+ scopes: Sequence[str],
+ relationship: str | None,
+ candidate_limit: int,
+ ) -> list[sqlite3.Row]:
+ """Indexed/coarse prefilter returning bounded canonical header rows."""
+ if candidate_limit <= 0:
+ raise ValueError("candidate limit must be positive")
+ columns = (
+ "id,type,subject,relation,value,scope,relationship,strength,"
+ "confidence,importance,updated_at"
+ )
+ per_bucket = max(32, candidate_limit // 4)
+ available_indexes = {
+ str(row[0]) for row in self._connection.execute(
+ "SELECT name FROM sqlite_master WHERE type='index'"
+ )
+ }
+ subject_hint = (
+ "INDEXED BY entry_subject_route"
+ if "entry_subject_route" in available_indexes else ""
+ )
+ scope_hint = (
+ "INDEXED BY entry_scope_route"
+ if "entry_scope_route" in available_indexes else ""
+ )
+ relationship_hint = (
+ "INDEXED BY entry_relationship_route"
+ if "entry_relationship_route" in available_indexes else ""
+ )
+ statements: list[tuple[str, tuple[object, ...]]] = [(
+ f"""SELECT {columns} FROM entries {subject_hint}
+ WHERE status=? AND subject=?
+ ORDER BY importance DESC,confidence DESC,id LIMIT ?""",
+ (PersonalityStatus.ACTIVE.value, subject, per_bucket),
+ )]
+ for scope in dict.fromkeys((*scopes, "global")):
+ statements.append((
+ f"""SELECT {columns} FROM entries {scope_hint}
+ WHERE status=? AND scope=?
+ ORDER BY importance DESC,confidence DESC,id LIMIT ?""",
+ (PersonalityStatus.ACTIVE.value, scope, per_bucket),
+ ))
+ if relationship is not None:
+ statements.append((
+ f"""SELECT {columns} FROM entries {relationship_hint}
+ WHERE status=? AND relationship=?
+ ORDER BY importance DESC,confidence DESC,id LIMIT ?""",
+ (PersonalityStatus.ACTIVE.value, relationship, per_bucket),
+ ))
+ rows: dict[str, sqlite3.Row] = {}
+ for statement, parameters in statements:
+ for row in self._connection.execute(statement, parameters):
+ rows.setdefault(str(row[0]), row)
+ return list(rows.values())
+
+ def _entry_json(self, entry: PersonalityEntry | None) -> str | None:
+ return None if entry is None else _canonical_json(asdict(entry))
+
+ def _upsert_entry(
+ self,
+ entry: PersonalityEntry,
+ *,
+ transaction_id: str,
+ action: str,
+ previous: PersonalityEntry | None,
+ ) -> None:
+ self._connection.execute(
+ f"INSERT OR REPLACE INTO entries({ENTRY_COLUMNS}) VALUES ({','.join('?' for _ in range(20))})",
+ _entry_to_row(entry),
+ )
+ self._connection.execute(
+ """INSERT INTO changes(transaction_id,timestamp,action,entry_id,before_json,after_json)
+ VALUES (?,?,?,?,?,?)""",
+ (
+ transaction_id, entry.updated_at, action, entry.id,
+ self._entry_json(previous), self._entry_json(entry),
+ ),
+ )
+
+ def ingest(
+ self,
+ record: EvidenceRecord,
+ *,
+ policy: PromotionPolicy = PromotionPolicy(),
+ importance: float = 0.5,
+ ) -> PromotionDecision:
+ """Persist evidence, then deterministically recompute its candidate family."""
+ self._writable()
+ try:
+ self._evidence_rows(record)
+ except sqlite3.IntegrityError as error:
+ self._connection.rollback()
+ self._readonly()
+ raise ValueError(f"duplicate evidence id: {record.id}") from error
+ family = self._candidate_records(record)
+ support = [item for item in family if item.value == record.value and item.polarity > 0]
+ direct_negative = [
+ item for item in family if item.value == record.value and item.polarity < 0
+ ]
+ opposing = [
+ item for item in family if item.value != record.value and item.polarity > 0
+ ] + direct_negative
+ score = policy.score(support)
+ opposing_weight = sum(
+ item.confidence * EvidenceAuthority(item.source_authority).weight
+ for item in opposing
+ )
+ net_score = max(0.0, score["promotion_score"] - policy.contradiction_weight * opposing_weight)
+ authorities = {EvidenceAuthority(item.source_authority) for item in support}
+ explicit_override = any(
+ EvidenceAuthority(item.source_authority) == EvidenceAuthority.EXPLICIT_USER
+ and item.confidence >= policy.explicit_override_confidence
+ for item in support
+ )
+ inference_only = bool(authorities) and authorities <= {
+ EvidenceAuthority.MODEL_INFERENCE, EvidenceAuthority.MODEL_UNSUPPORTED
+ }
+ promotable = (
+ bool(support)
+ and not inference_only
+ and (
+ explicit_override
+ or (len(support) >= 3 and net_score >= policy.promotion_threshold)
+ )
+ )
+ entry_id = _stable_id(
+ "personality", record.entry_type, record.subject, record.relation,
+ record.value, record.scope, record.relationship,
+ )
+ transaction_id = _stable_id("change", record.id, record.timestamp)
+ existing_row = self._connection.execute(
+ f"SELECT {ENTRY_COLUMNS} FROM entries WHERE id=?", (entry_id,)
+ ).fetchone()
+ existing = None if existing_row is None else _row_to_entry(tuple(existing_row))
+ action = "evidence_only"
+ reason = "promotion threshold not reached"
+ promoted_entry_id: str | None = None
+
+ if promotable:
+ weighted_support = score["weighted_support"]
+ confidence = weighted_support / (
+ weighted_support + opposing_weight + policy.confidence_prior
+ )
+ if explicit_override:
+ confidence = max(confidence, 0.9)
+ strength = min(1.0, net_score / (2 * policy.promotion_threshold))
+ strongest = max(
+ support,
+ key=lambda item: (
+ EvidenceAuthority(item.source_authority).weight, item.confidence
+ ),
+ )
+ created_at = existing.created_at if existing else record.timestamp
+ entry = PersonalityEntry(
+ id=entry_id, entry_type=record.entry_type, subject=record.subject,
+ relation=record.relation, value=record.value, scope=record.scope,
+ relationship=record.relationship, strength=strength,
+ confidence=confidence, importance=importance,
+ evidence_count=len(support),
+ context_diversity=int(score["context_diversity"]),
+ last_reinforced=max(item.timestamp for item in support),
+ created_at=created_at, updated_at=record.timestamp,
+ source_authority=strongest.source_authority,
+ supporting_evidence_ids=tuple(sorted(item.id for item in support)),
+ contradicting_evidence_ids=tuple(sorted(item.id for item in opposing)),
+ status=PersonalityStatus.ACTIVE.value,
+ extension=(existing.extension if existing else {}),
+ )
+ # A correction never erases the old conclusion; it supersedes it.
+ active_opponents = self._connection.execute(
+ f"""SELECT {ENTRY_COLUMNS} FROM entries WHERE type=? AND subject=?
+ AND relation=? AND scope=? AND relationship IS ? AND status=? AND id != ?""",
+ (
+ record.entry_type, record.subject, record.relation, record.scope,
+ record.relationship, PersonalityStatus.ACTIVE.value, entry_id,
+ ),
+ ).fetchall()
+ for opponent_row in active_opponents:
+ opponent = _row_to_entry(tuple(opponent_row))
+ new_status = (
+ PersonalityStatus.SUPERSEDED.value
+ if explicit_override else PersonalityStatus.CONTRADICTED.value
+ )
+ changed = replace(
+ opponent,
+ confidence=max(0.0, opponent.confidence - opposing_weight / (1 + opposing_weight)),
+ contradicting_evidence_ids=tuple(sorted(set(
+ opponent.contradicting_evidence_ids + tuple(item.id for item in support)
+ ))),
+ status=new_status, updated_at=record.timestamp,
+ )
+ self._upsert_entry(
+ changed, transaction_id=transaction_id,
+ action="supersede" if explicit_override else "contradict",
+ previous=opponent,
+ )
+ self._upsert_entry(
+ entry, transaction_id=transaction_id,
+ action="create" if existing is None else "reinforce", previous=existing,
+ )
+ action = "create" if existing is None else "reinforce"
+ reason = "explicit authority override" if explicit_override else "promotion threshold reached"
+ promoted_entry_id = entry.id
+ elif existing is not None and opposing:
+ lowered = replace(
+ existing,
+ confidence=max(0.0, existing.confidence - opposing_weight / (1 + opposing_weight)),
+ contradicting_evidence_ids=tuple(sorted(set(
+ existing.contradicting_evidence_ids + tuple(item.id for item in opposing)
+ ))),
+ status=(
+ PersonalityStatus.CONTRADICTED.value
+ if existing.confidence < 0.5 else existing.status
+ ),
+ updated_at=record.timestamp,
+ )
+ self._upsert_entry(
+ lowered, transaction_id=transaction_id, action="lower_confidence",
+ previous=existing,
+ )
+ action = "lower_confidence"
+ reason = "contradictory evidence recorded"
+ promoted_entry_id = existing.id
+ self._mark_dirty(record.timestamp)
+ self._connection.commit()
+ self._readonly()
+ return PromotionDecision(
+ evidence_id=record.id, promoted=promotable,
+ entry_id=promoted_entry_id, action=action,
+ promotion_score=net_score, threshold=policy.promotion_threshold,
+ reason=reason,
+ )
+
+ def changes(self) -> list[dict[str, object]]:
+ return [dict(row) for row in self._connection.execute(
+ "SELECT * FROM changes ORDER BY sequence"
+ )]
+
+ def undo_last(self, *, timestamp: str | None = None) -> str:
+ last = self._connection.execute(
+ "SELECT transaction_id FROM changes ORDER BY sequence DESC LIMIT 1"
+ ).fetchone()
+ if last is None:
+ raise ValueError("personality package has no reversible changes")
+ transaction_id = str(last[0])
+ rows = self._connection.execute(
+ "SELECT * FROM changes WHERE transaction_id=? ORDER BY sequence DESC",
+ (transaction_id,),
+ ).fetchall()
+ self._writable()
+ for row in rows:
+ before = row[5]
+ if before is None:
+ self._connection.execute("DELETE FROM entries WHERE id=?", (row[4],))
+ else:
+ data = json.loads(before)
+ data["supporting_evidence_ids"] = tuple(data["supporting_evidence_ids"])
+ data["contradicting_evidence_ids"] = tuple(data["contradicting_evidence_ids"])
+ restored = PersonalityEntry(**data)
+ self._connection.execute(
+ f"INSERT OR REPLACE INTO entries({ENTRY_COLUMNS}) VALUES ({','.join('?' for _ in range(20))})",
+ _entry_to_row(restored),
+ )
+ self._connection.execute("DELETE FROM changes WHERE transaction_id=?", (transaction_id,))
+ self._mark_dirty(timestamp or _utc_now())
+ self._connection.commit()
+ self._readonly()
+ return transaction_id
+
+ def bulk_insert_entries(
+ self, entries: Iterable[PersonalityEntry], *, updated_at: str
+ ) -> int:
+ """Benchmark/import path; it never bypasses canonical schema validation."""
+ rows = []
+ for entry in entries:
+ PersonalityType(entry.entry_type)
+ PersonalityStatus(entry.status)
+ EvidenceAuthority(entry.source_authority)
+ rows.append(_entry_to_row(entry))
+ self._writable()
+ self._connection.executemany(
+ f"INSERT INTO entries({ENTRY_COLUMNS}) VALUES ({','.join('?' for _ in range(20))})",
+ rows,
+ )
+ self._mark_dirty(updated_at)
+ self._connection.commit()
+ self._readonly()
+ return len(rows)
+
+
+class PersonalityRouter:
+ """Canonical CPU/disk router; no model hidden width or CUDA state."""
+
+ def __init__(
+ self,
+ *,
+ acceptance_threshold: float = 5.0,
+ candidate_limit: int = 512,
+ entity_width: int = 128,
+ ) -> None:
+ self.acceptance_threshold = acceptance_threshold
+ self.candidate_limit = candidate_limit
+ self.encoder = ByteEntityEncoder(entity_width)
+
+ @staticmethod
+ def _header_bytes(rows: Sequence[sqlite3.Row]) -> int:
+ return sum(len(_canonical_json(tuple(row)).encode()) for row in rows)
+
+ def route(
+ self, package: PersonalityPackage, query: PersonalityQuery, *, top_k: int = 4
+ ) -> PersonalityRoute:
+ if top_k not in (1, 4, 8):
+ raise ValueError("P-package router supports top_k 1, 4, or 8")
+ rows = package.routing_headers(
+ subject=query.subject,
+ scopes=(query.interaction_type, query.domain),
+ relationship=query.relationship,
+ candidate_limit=self.candidate_limit,
+ )
+ if not rows:
+ return PersonalityRoute((), (), 0, 0, 0)
+ query_semantic = self.encoder.encode_one(
+ " ".join(filter(None, (
+ query.subject, query.relation or "", query.interaction_type,
+ query.domain, query.relationship or "",
+ )))
+ )
+ now = _parse_time(query.timestamp) if query.timestamp else datetime.now(timezone.utc)
+ scored: list[tuple[float, str]] = []
+ for row in rows:
+ semantic = float(F.cosine_similarity(
+ query_semantic,
+ self.encoder.encode_one(" ".join((row[2], row[3], row[4]))),
+ dim=0,
+ ))
+ identity = 1.0 if row[2].casefold() == query.subject.casefold() else 0.15
+ if row[5] == query.interaction_type:
+ context = 1.0
+ elif row[5] == query.domain:
+ context = 0.85
+ elif row[5] == "global":
+ context = 0.4
+ else:
+ context = 0.05
+ if row[6] is None:
+ relationship = 0.35
+ elif query.relationship and row[6].casefold() == query.relationship.casefold():
+ relationship = 1.0
+ else:
+ relationship = 0.0
+ relation = 0.5
+ if query.relation:
+ relation = 1.0 if row[3].casefold() == query.relation.casefold() else 0.0
+ age_days = max(0.0, (now - _parse_time(row[10])).total_seconds() / 86400)
+ freshness = math.exp(-age_days / 3650)
+ score = (
+ 1.5 * semantic + 2.0 * identity + 2.0 * context
+ + 1.5 * relationship + relation + float(row[7])
+ + float(row[8]) + float(row[9]) + 0.25 * freshness
+ )
+ scored.append((score, str(row[0])))
+ scored.sort(key=lambda item: (-item[0], item[1]))
+ accepted = [item for item in scored if item[0] >= self.acceptance_threshold][:top_k]
+ return PersonalityRoute(
+ entry_ids=tuple(item[1] for item in accepted),
+ scores=tuple(item[0] for item in accepted),
+ candidate_count=len(rows), header_bytes_read=self._header_bytes(rows),
+ accepted=len(accepted),
+ )
+
+ def retrieve(
+ self, package: PersonalityPackage, query: PersonalityQuery, *, top_k: int = 4
+ ) -> PersonalitySelection:
+ import time
+
+ started = time.perf_counter()
+ route = self.route(package, query, top_k=top_k)
+ entries = tuple(package.entry(entry_id) for entry_id in route.entry_ids)
+ entry_bytes = sum(len(_canonical_json(asdict(entry)).encode()) for entry in entries)
+ return PersonalitySelection(
+ entries=entries, route=route, entry_bytes_read=entry_bytes,
+ retrieval_latency_seconds=time.perf_counter() - started,
+ )
+
+
+class FactorizedPersonalityCanonicalizer:
+ """Model-independent personality entry -> existing canonical-P protocol."""
+
+ def __init__(
+ self,
+ representation: FactorizedStateRepresentation,
+ *,
+ value_labels: Sequence[str],
+ ) -> None:
+ self.representation = representation.cpu().eval()
+ self.value_labels = tuple(value_labels)
+
+ @staticmethod
+ def _index(value: str, size: int) -> int:
+ return int.from_bytes(hashlib.sha256(value.casefold().encode()).digest()[:8], "big") % size
+
+ def ids(self, entry: PersonalityEntry) -> tuple[int, int, int, int]:
+ value_lookup = {label.casefold(): index for index, label in enumerate(self.value_labels)}
+ relation_aliases = {
+ "owner": 0,
+ "preferred_persona": 0,
+ "response_style": 0,
+ "location": 1,
+ "status": 2,
+ }
+ return (
+ self._index(entry.subject, self.representation.entity.num_embeddings),
+ relation_aliases.get(
+ entry.relation.casefold(),
+ self._index(entry.relation, self.representation.relation.num_embeddings),
+ ),
+ value_lookup.get(
+ entry.value.casefold(),
+ self._index(entry.value, self.representation.value.num_embeddings),
+ ),
+ CANONICAL,
+ )
+
+ def encode(self, entry: PersonalityEntry) -> tuple[Tensor, tuple[int, int, int, int]]:
+ ids = self.ids(entry)
+ fields = [torch.tensor([value]) for value in ids]
+ with torch.inference_mode():
+ vector = self.representation.encode(*fields)[0].detach().cpu()
+ return vector, ids
+
+
+@dataclass(frozen=True)
+class PersonalityActivation:
+ selection: PersonalitySelection
+ store: CanonicalPStore | None
+ canonical_bytes: int
+ inactive_vram_bytes: int = 0
+
+
+class PersonalityTranslateSession:
+ """Validate once, then reuse one read connection for bounded top-k activation."""
+
+ def __init__(
+ self,
+ package_path: str | Path,
+ router: PersonalityRouter,
+ canonicalizer: FactorizedPersonalityCanonicalizer,
+ *,
+ validate_on_open: bool = True,
+ ) -> None:
+ self.package_path = Path(package_path)
+ self.router = router
+ self.canonicalizer = canonicalizer
+ self.validated_checksum: str | None = None
+ self._validated_stat: tuple[int, int] | None = None
+ if validate_on_open:
+ with PersonalityPackage(self.package_path, validate=True) as package:
+ self.validated_checksum = package.metadata()["content_sha256"]
+ stat = self.package_path.stat()
+ self._validated_stat = (stat.st_size, stat.st_mtime_ns)
+ self._package = PersonalityPackage(self.package_path, validate=False)
+
+ def close(self) -> None:
+ self._package.close(checkpoint=False)
+
+ def __enter__(self) -> "PersonalityTranslateSession":
+ return self
+
+ def __exit__(self, *_exc) -> None:
+ self.close()
+
+ def activate(
+ self,
+ query: PersonalityQuery,
+ *,
+ top_k: int = 4,
+ device: str | torch.device = "cpu",
+ dtype: torch.dtype = torch.float16,
+ ) -> PersonalityActivation:
+ if self._validated_stat is not None:
+ stat = self.package_path.stat()
+ if (stat.st_size, stat.st_mtime_ns) != self._validated_stat:
+ raise ValueError("personality package changed after integrity validation")
+ # Full semantic validation is a cold-open operation. Per-generation
+ # activation reuses the read connection and reads bounded router headers
+ # plus selected full rows.
+ selection = self.router.retrieve(self._package, query, top_k=top_k)
+ if not selection.entries:
+ return PersonalityActivation(selection, None, 0)
+ capacity = len(selection.entries)
+ store = CanonicalPStore(CanonicalPConfig(
+ slots=capacity, width=512, dtype=dtype, device=device,
+ merge_similarity=1.0,
+ ))
+ for entry in selection.entries:
+ vector, ids = self.canonicalizer.encode(entry)
+ store.create(
+ vector, entity_id=ids[0], relation_id=ids[1], value_id=ids[2],
+ metadata_id=ids[3], slot_type=SlotType.FACT,
+ confidence=entry.confidence, importance=entry.importance,
+ freshness=Freshness.FRESH, persistence=Persistence.DURABLE,
+ source=SlotSource.CONVERSATION, label=entry.subject,
+ )
+ tensors = (
+ store.cache.values, store.cache.valid, store.cache.slot_type,
+ store.cache.confidence, store.cache.importance, store.cache.freshness,
+ store.cache.persistence, store.cache.last_updated, store.cache.source,
+ store.entity_id, store.relation_id, store.value_id,
+ store.canonical_metadata_id,
+ )
+ canonical_bytes = sum(tensor.numel() * tensor.element_size() for tensor in tensors)
+ return PersonalityActivation(selection, store, canonical_bytes)
+
+
+def merge_active_personality_with_p_cache(
+ p_cache: CanonicalPStore | None, personality: CanonicalPStore | None
+) -> CanonicalPStore:
+ """Construct a temporary combined view without mutating either sibling."""
+ p_count = 0 if p_cache is None else p_cache.cache.occupied
+ personality_count = 0 if personality is None else personality.cache.occupied
+ capacity = max(1, p_count + personality_count)
+ if personality is not None:
+ device = personality.cache.device
+ dtype = personality.cache.values.dtype
+ elif p_cache is not None:
+ device = p_cache.cache.device
+ dtype = p_cache.cache.values.dtype
+ else:
+ device = torch.device("cpu")
+ dtype = torch.float16
+ merged = CanonicalPStore(CanonicalPConfig(
+ slots=capacity, width=512, dtype=dtype,
+ device=device, merge_similarity=1.0,
+ ))
+ for source in (p_cache, personality):
+ if source is None:
+ continue
+ for index in source.valid.nonzero(as_tuple=False).flatten().tolist():
+ merged.create(
+ source.canonical_values[index], entity_id=int(source.entity_id[index]),
+ relation_id=int(source.relation_id[index]), value_id=int(source.value_id[index]),
+ metadata_id=int(source.canonical_metadata_id[index]),
+ slot_type=SlotType(int(source.cache.slot_type[index])),
+ confidence=float(source.cache.confidence[index]),
+ importance=float(source.cache.importance[index]),
+ freshness=Freshness(int(source.cache.freshness[index])),
+ persistence=Persistence(int(source.cache.persistence[index])),
+ source=SlotSource(int(source.cache.source[index])),
+ label=source.cache.labels[index],
+ )
+ return merged
+
+
+def evidence_from_p_cache(
+ store: CanonicalPStore,
+ slot: int,
+ *,
+ evidence_id: str,
+ entry_type: str,
+ relation: str,
+ value: str,
+ context: str,
+ scope: str,
+ timestamp: str,
+ archive_reference: str,
+ behavioral: bool,
+ relationship: str | None = None,
+) -> EvidenceRecord | None:
+ """Explicitly gated P->evidence flow; ordinary mutable facts return None."""
+ if not behavioral or not bool(store.valid[slot]):
+ return None
+ return EvidenceRecord(
+ id=evidence_id, entry_type=entry_type,
+ subject=store.cache.labels[slot] or f"entity:{int(store.entity_id[slot])}",
+ relation=relation, value=value, context=context, scope=scope,
+ confidence=float(store.cache.confidence[slot]),
+ source_authority=EvidenceAuthority.SINGLE_OBSERVED.value,
+ timestamp=timestamp, archive_reference=archive_reference,
+ relationship=relationship,
+ )
+
+
+def synthetic_entry(
+ index: int,
+ *,
+ timestamp: str = "2026-01-01T00:00:00+00:00",
+) -> PersonalityEntry:
+ """Deterministic growth-benchmark fixture."""
+ evidence_id = f"archive-evidence-{index}"
+ return PersonalityEntry(
+ id=f"personality-{index:08d}",
+ entry_type=PersonalityType.CONTEXTUAL_TENDENCY.value,
+ subject=f"subject-{index % 4096}", relation=f"trait-{index % 97}",
+ value=f"value-{index}", scope=f"domain-{index % 31}",
+ relationship=None, strength=0.75, confidence=0.8,
+ importance=0.5, evidence_count=3, context_diversity=2,
+ last_reinforced=timestamp, created_at=timestamp, updated_at=timestamp,
+ source_authority=EvidenceAuthority.REPEATED_OBSERVED.value,
+ supporting_evidence_ids=(evidence_id,), contradicting_evidence_ids=(),
+ )
diff --git a/src/pcm/planner/personality_eval.py b/src/pcm/planner/personality_eval.py
new file mode 100644
index 0000000000000000000000000000000000000000..bf8f7a916c34875b196c7b6dd12bfe21bd7ce705
--- /dev/null
+++ b/src/pcm/planner/personality_eval.py
@@ -0,0 +1,669 @@
+"""Reproducible mechanical and CUDA evaluation for `.ppkg` v1."""
+
+from __future__ import annotations
+
+from dataclasses import asdict
+import gc
+import hashlib
+import json
+import math
+from pathlib import Path
+import random
+import tempfile
+import time
+import tracemalloc
+
+import torch
+import torch.nn.functional as F
+from transformers import AutoModelForCausalLM, AutoTokenizer
+
+from pcm.planner.canonical import CanonicalPConfig, CanonicalPStore
+from pcm.planner.personality import (
+ EvidenceAuthority,
+ EvidenceRecord,
+ FactorizedPersonalityCanonicalizer,
+ PersonalityPackage,
+ PersonalityQuery,
+ PersonalityRouter,
+ PersonalityStatus,
+ PersonalityTranslateSession,
+ PersonalityType,
+ PromotionPolicy,
+ merge_active_personality_with_p_cache,
+ synthetic_entry,
+)
+from pcm.planner.pythia_split_translate import PythiaSplitTranslatedModel
+from pcm.planner.representation import FactorizedStateRepresentation
+from pcm.planner.compatibility import TensorTranslationLayer
+from pcm.planner.split_translator import ByteEntityEncoder, CanonicalPRouter
+from pcm.planner.canonical import CANONICAL_VALUE_LABELS as VALUE_LABELS
+
+
+STAMP = "2026-06-01T00:00:00+00:00"
+NATURAL_RP = (
+ "Mira folded the map, squinted at the crooked ink, and laughed.\nCaptain:",
+ "Rain tapped the observatory windows while the old astronomer adjusted the brass lens.\nGuest:",
+ "The fox-eared courier set down her tea and nudged the sealed letter across the table.\nCourier:",
+)
+
+
+def _file_sha256(path: Path) -> str:
+ digest = hashlib.sha256()
+ with path.open("rb") as handle:
+ for chunk in iter(lambda: handle.read(1024 * 1024), b""):
+ digest.update(chunk)
+ return digest.hexdigest()
+
+
+def _evidence(
+ prefix: str,
+ index: int,
+ *,
+ value: str,
+ context: str,
+ scope: str,
+ authority: EvidenceAuthority = EvidenceAuthority.SINGLE_OBSERVED,
+ confidence: float = 0.8,
+ relation: str = "response_style",
+ subject: str = "user",
+ relationship: str | None = None,
+ connected: tuple[str, ...] = (),
+) -> EvidenceRecord:
+ return EvidenceRecord(
+ id=f"{prefix}-{index}", entry_type=PersonalityType.INTERACTION_STYLE.value,
+ subject=subject, relation=relation, value=value, context=context,
+ scope=scope, confidence=confidence, source_authority=authority.value,
+ timestamp=f"2026-06-{index + 1:02d}T00:00:00+00:00",
+ archive_reference=f"archive://heldout/{prefix}/{index}",
+ relationship=relationship, connected_evidence_ids=connected,
+ )
+
+
+def _promote(
+ package: PersonalityPackage,
+ prefix: str,
+ *,
+ value: str,
+ scope: str,
+ relation: str = "response_style",
+ subject: str = "user",
+ contexts: tuple[str, ...] = ("debugging", "planning", "explanation"),
+ relationship: str | None = None,
+):
+ decision = None
+ for index, context in enumerate(contexts):
+ decision = package.ingest(_evidence(
+ prefix, index, value=value, context=context, scope=scope,
+ relation=relation, subject=subject, relationship=relationship,
+ ))
+ if decision is None or not decision.promoted:
+ raise RuntimeError("synthetic durable personality did not promote")
+ return package.entry(decision.entry_id), decision
+
+
+def build_proof_package(path: str | Path) -> dict[str, object]:
+ path = Path(path)
+ policy = PromotionPolicy()
+ with PersonalityPackage.create(
+ path, package_id="phase-b-personality-proof", created_at=STAMP, overwrite=True,
+ metadata={"purpose": "mechanical durable personality proof"},
+ ) as package:
+ weak = package.ingest(_evidence(
+ "weak", 0, value="verbose", context="one-off", scope="global",
+ authority=EvidenceAuthority.SINGLE_OBSERVED,
+ ))
+ repeated, repeated_decision = _promote(
+ package, "repeat", value="concise", scope="global"
+ )
+ narrow = [
+ _evidence(
+ "narrow", index, value="structured", context="debugging", scope="global",
+ authority=EvidenceAuthority.SINGLE_OBSERVED,
+ )
+ for index in range(5)
+ ]
+ connected = [
+ _evidence(
+ "connected", index, value="direct", context=context, scope="global",
+ authority=EvidenceAuthority.SINGLE_OBSERVED,
+ connected=((f"connected-{index - 1}",) if index else ()),
+ )
+ for index, context in enumerate(("debugging", "planning", "casual"))
+ ]
+ narrow_score = policy.score(narrow)["promotion_score"]
+ connected_score = policy.score(connected)["promotion_score"]
+ unsupported = None
+ for index in range(20):
+ unsupported = package.ingest(_evidence(
+ "unsupported", index, value="hostile", context=f"self-{index}",
+ scope="global", authority=EvidenceAuthority.MODEL_UNSUPPORTED,
+ ))
+ correction = package.ingest(_evidence(
+ "correction", 20, value="detailed", context="explicit-correction",
+ scope="global", authority=EvidenceAuthority.EXPLICIT_USER,
+ confidence=1.0,
+ ))
+ technical, _ = _promote(
+ package, "technical", value="concise", scope="technical"
+ )
+ creative, _ = _promote(
+ package, "creative", value="detailed", scope="creative",
+ contexts=("fiction", "poetry", "worldbuilding"),
+ )
+ relationship, _ = _promote(
+ package, "relationship", value="playful", scope="roleplay",
+ subject="assistant", relationship="captain-mira",
+ )
+ persona, _ = _promote(
+ package, "persona", value="Alice", scope="technical",
+ relation="preferred_persona",
+ )
+ router = PersonalityRouter()
+ technical_route = router.retrieve(package, PersonalityQuery(
+ subject="user", interaction_type="technical", domain="debugging",
+ relation="response_style", timestamp=STAMP,
+ ), top_k=1)
+ creative_route = router.retrieve(package, PersonalityQuery(
+ subject="user", interaction_type="creative", domain="fiction",
+ relation="response_style", timestamp=STAMP,
+ ), top_k=1)
+ relationship_route = router.retrieve(package, PersonalityQuery(
+ subject="assistant", interaction_type="roleplay", domain="chat",
+ relation="response_style", relationship="captain-mira", timestamp=STAMP,
+ ), top_k=1)
+ irrelevant = router.retrieve(package, PersonalityQuery(
+ subject="stranger", interaction_type="medical", domain="finance",
+ relation="unrelated", timestamp=STAMP,
+ ), top_k=8)
+ topk = {}
+ for count in (1, 4, 8):
+ result = router.retrieve(package, PersonalityQuery(
+ subject="user", interaction_type="technical", domain="debugging",
+ relation="response_style", timestamp=STAMP,
+ ), top_k=count)
+ topk[str(count)] = {
+ "loaded_entries": len(result.entries),
+ "target_recall": float(technical.id in result.route.entry_ids),
+ "logical_bytes_read": result.logical_bytes_read,
+ "latency_seconds": result.retrieval_latency_seconds,
+ }
+ package.checkpoint()
+ return {
+ "one_event_promoted": weak.promoted,
+ "repeated_promoted": repeated_decision.promoted,
+ "repeated_entry": asdict(repeated),
+ "narrow_context_score": narrow_score,
+ "connected_cross_context_score": connected_score,
+ "connectivity_outscores_narrow": connected_score > narrow_score,
+ "unsupported_model_claim_promoted": bool(unsupported and unsupported.promoted),
+ "explicit_correction_promoted": correction.promoted,
+ "old_conclusion_status": package.entry(repeated.id).status,
+ "technical_context_correct": technical_route.entries[0].id == technical.id,
+ "creative_context_correct": creative_route.entries[0].id == creative.id,
+ "relationship_context_correct": relationship_route.entries[0].id == relationship.id,
+ "relevance_accuracy": sum((
+ technical_route.entries[0].id == technical.id,
+ creative_route.entries[0].id == creative.id,
+ relationship_route.entries[0].id == relationship.id,
+ len(irrelevant.entries) == 0,
+ )) / 4,
+ "irrelevant_loaded_entries": len(irrelevant.entries),
+ "top_k": topk,
+ "persona_entry_id": persona.id,
+ "entry_count": len(package.entries()),
+ "change_count": len(package.changes()),
+ "package_size_bytes": package.size_bytes,
+ "semantic_checksum": package.metadata()["content_sha256"],
+ }
+
+
+def deterministic_serialization_proof(root: Path) -> dict[str, object]:
+ first = root / "deterministic-a.ppkg"
+ second = root / "deterministic-b.ppkg"
+ build_proof_package(first)
+ build_proof_package(second)
+ first_hash = _file_sha256(first)
+ second_hash = _file_sha256(second)
+ return {
+ "byte_identical": first.read_bytes() == second.read_bytes(),
+ "first_sha256": first_hash,
+ "second_sha256": second_hash,
+ }
+
+
+def growth_benchmark(root: Path, counts=(100, 1_000, 10_000, 100_000)):
+ results = {}
+ router = PersonalityRouter()
+ for count in counts:
+ path = root / f"growth-{count}.ppkg"
+ with PersonalityPackage.create(
+ path, package_id=f"growth-{count}", created_at=STAMP, overwrite=True,
+ ) as package:
+ package.bulk_insert_entries(
+ (synthetic_entry(index, timestamp=STAMP) for index in range(count)),
+ updated_at=STAMP,
+ )
+ tracemalloc.start()
+ validation_started = time.perf_counter()
+ with PersonalityPackage(path) as package:
+ validation_latency = time.perf_counter() - validation_started
+ before_current, before_peak = tracemalloc.get_traced_memory()
+ selection = router.retrieve(package, PersonalityQuery(
+ subject="subject-0", interaction_type="domain-0", domain="domain-0",
+ relation="trait-0", timestamp=STAMP,
+ ), top_k=4)
+ after_current, after_peak = tracemalloc.get_traced_memory()
+ size = package.size_bytes
+ tracemalloc.stop()
+ results[str(count)] = {
+ "disk_size_bytes": size,
+ "lookup_latency_seconds": selection.retrieval_latency_seconds,
+ "cold_checksum_validation_seconds": validation_latency,
+ "loaded_entries": len(selection.entries),
+ "logical_bytes_read": selection.logical_bytes_read,
+ "python_ram_peak_delta_bytes": max(0, after_peak - before_peak),
+ "inactive_vram_bytes": 0,
+ "active_canonical_bytes": len(selection.entries) * 1077,
+ "full_package_loaded": False,
+ }
+ path.unlink()
+ return results
+
+
+def _entry_store_bytes(store: CanonicalPStore | None) -> int:
+ if store is None:
+ return 0
+ tensors = (
+ store.cache.values, store.cache.valid, store.cache.slot_type,
+ store.cache.confidence, store.cache.importance, store.cache.freshness,
+ store.cache.persistence, store.cache.last_updated, store.cache.source,
+ store.entity_id, store.relation_id, store.value_id,
+ store.canonical_metadata_id,
+ )
+ return sum(value.numel() * value.element_size() for value in tensors)
+
+
+def _make_persona_package(path: Path, value: str) -> None:
+ with PersonalityPackage.create(
+ path, package_id=f"persona-{value.casefold()}", created_at=STAMP, overwrite=True,
+ ) as package:
+ _promote(
+ package, "persona", value=value, scope="technical",
+ relation="preferred_persona",
+ )
+
+
+def cuda_ttl_benchmark(
+ *,
+ model_path: Path,
+ ttl_path: Path,
+ router_path: Path,
+ proof_package_path: Path,
+ representation: FactorizedStateRepresentation,
+ workdir: Path,
+) -> dict[str, object]:
+ if not torch.cuda.is_available():
+ raise RuntimeError("CUDA is required for exact `.translate` personality proof")
+ tokenizer = AutoTokenizer.from_pretrained(model_path, local_files_only=True)
+ tokenizer.pad_token = tokenizer.eos_token
+ tokenizer.padding_side = "left"
+ base = AutoModelForCausalLM.from_pretrained(
+ model_path, local_files_only=True, dtype=torch.float16, low_cpu_mem_usage=True,
+ ).to("cuda").eval()
+ package = TensorTranslationLayer.load(ttl_path, device="cuda")
+ canonical_router = CanonicalPRouter.load(router_path, device="cuda")
+ wrapper = PythiaSplitTranslatedModel(
+ base, package, canonical_router, ByteEntityEncoder(128)
+ ).to("cuda").eval()
+ canonicalizer = FactorizedPersonalityCanonicalizer(
+ representation, value_labels=VALUE_LABELS
+ )
+ disk_router = PersonalityRouter()
+ package_a = workdir / "persona-a.ppkg"
+ package_b = workdir / "persona-b.ppkg"
+ package_irrelevant = workdir / "persona-irrelevant.ppkg"
+ package_low = workdir / "persona-low-confidence.ppkg"
+ package_context = workdir / "persona-context.ppkg"
+ _make_persona_package(package_a, "Alice")
+ _make_persona_package(package_b, "Bob")
+ with PersonalityPackage.create(
+ package_irrelevant, package_id="persona-irrelevant", created_at=STAMP,
+ overwrite=True,
+ ) as irrelevant_package:
+ _promote(
+ irrelevant_package, "other", value="Alice", scope="creative",
+ relation="preferred_persona", subject="someone-else",
+ )
+ with PersonalityPackage.create(
+ package_low, package_id="persona-low", created_at=STAMP, overwrite=True,
+ ) as low_package:
+ low_package.ingest(_evidence(
+ "low", 0, value="Alice", context="debugging", scope="technical",
+ relation="preferred_persona", authority=EvidenceAuthority.SINGLE_OBSERVED,
+ confidence=0.3,
+ ))
+ with PersonalityPackage.create(
+ package_context, package_id="persona-context", created_at=STAMP, overwrite=True,
+ ) as context_package:
+ _promote(
+ context_package, "context-technical", value="Alice", scope="technical",
+ relation="preferred_persona",
+ )
+ _promote(
+ context_package, "context-creative", value="Bob", scope="creative",
+ relation="preferred_persona", contexts=("fiction", "poetry", "worldbuilding"),
+ )
+ query = PersonalityQuery(
+ subject="user", interaction_type="technical", domain="debugging",
+ relation="preferred_persona", timestamp=STAMP,
+ )
+ before_inactive = torch.cuda.memory_allocated()
+ with PersonalityPackage(proof_package_path) as disk_only:
+ inactive_entries = int(disk_only._connection.execute(
+ "SELECT COUNT(*) FROM entries WHERE status=?",
+ (PersonalityStatus.ACTIVE.value,),
+ ).fetchone()[0])
+ after_inactive = torch.cuda.memory_allocated()
+ sessions = {
+ "package_a": PersonalityTranslateSession(package_a, disk_router, canonicalizer),
+ "package_b": PersonalityTranslateSession(package_b, disk_router, canonicalizer),
+ "irrelevant": PersonalityTranslateSession(package_irrelevant, disk_router, canonicalizer),
+ "low_confidence": PersonalityTranslateSession(package_low, disk_router, canonicalizer),
+ }
+ activations = {
+ label: session.activate(query, top_k=1, device="cuda")
+ for label, session in sessions.items()
+ }
+ context_session = PersonalityTranslateSession(
+ package_context, disk_router, canonicalizer
+ )
+ activations["context_technical"] = context_session.activate(
+ query, top_k=1, device="cuda"
+ )
+ activations["context_creative"] = context_session.activate(
+ PersonalityQuery(
+ subject="user", interaction_type="creative", domain="fiction",
+ relation="preferred_persona", timestamp=STAMP,
+ ),
+ top_k=1, device="cuda",
+ )
+ prompt = "Following a fresh neutral technical exchange, The owner of the user is"
+ inputs = tokenizer(prompt, return_tensors="pt", add_special_tokens=False).to("cuda")
+ alice_id = tokenizer.encode(" Alice", add_special_tokens=False)[0]
+ bob_id = tokenizer.encode(" Bob", add_special_tokens=False)[0]
+
+ p_cache = CanonicalPStore(CanonicalPConfig(
+ slots=1, width=512, dtype=torch.float16, device="cuda", merge_similarity=1.0,
+ ))
+ other_entry = activations["package_a"].selection.entries[0]
+ other_entry = type(other_entry)(**{
+ **asdict(other_entry), "subject": "current-task", "value": "Alice",
+ "supporting_evidence_ids": tuple(other_entry.supporting_evidence_ids),
+ "contradicting_evidence_ids": tuple(other_entry.contradicting_evidence_ids),
+ })
+ vector, ids = canonicalizer.encode(other_entry)
+ p_cache.create(
+ vector, entity_id=ids[0], relation_id=ids[1], value_id=ids[2],
+ metadata_id=ids[3], label="current-task",
+ )
+ combined = merge_active_personality_with_p_cache(
+ p_cache, activations["package_a"].store
+ )
+
+ def altered_store(*, label="user", relation_id=None, metadata_id=None, invalidate=False):
+ source = activations["package_a"].store
+ source_slot = int(source.valid.nonzero(as_tuple=False)[0])
+ result = CanonicalPStore(CanonicalPConfig(
+ slots=1, width=512, dtype=torch.float16, device="cuda",
+ merge_similarity=1.0,
+ ))
+ slot, _ = result.create(
+ source.canonical_values[source_slot],
+ entity_id=int(source.entity_id[source_slot]),
+ relation_id=(
+ int(source.relation_id[source_slot])
+ if relation_id is None else relation_id
+ ),
+ value_id=int(source.value_id[source_slot]),
+ metadata_id=(
+ int(source.canonical_metadata_id[source_slot])
+ if metadata_id is None else metadata_id
+ ),
+ label=label,
+ )
+ if invalidate:
+ result.invalidate(slot)
+ return result
+
+ wrong_entity = altered_store(label="someone-else")
+ wrong_relation = altered_store(relation_id=1)
+ historical = altered_store(metadata_id=2)
+ invalidated = altered_store(invalidate=True)
+
+ conditions = {
+ "frozen_base": (None, {}),
+ "router_disabled": (None, {}),
+ "translator_disabled": (
+ activations["package_a"].store, {"injection_enabled": False}
+ ),
+ "translator_oracle_route": (
+ activations["package_a"].store,
+ {"oracle_indices": torch.tensor([0], device="cuda")},
+ ),
+ "p_cache_only": (p_cache, {}),
+ "wrong_entity": (wrong_entity, {}),
+ "wrong_relation": (wrong_relation, {}),
+ "historical": (historical, {}),
+ "invalidated": (invalidated, {}),
+ "p_package_a_relevant": (activations["package_a"].store, {}),
+ "p_package_b_relevant": (activations["package_b"].store, {}),
+ "p_package_irrelevant": (activations["irrelevant"].store, {}),
+ "p_package_contradictory_low_confidence": (
+ activations["low_confidence"].store, {}
+ ),
+ "p_package_context_technical": (activations["context_technical"].store, {}),
+ "p_package_context_creative": (activations["context_creative"].store, {}),
+ "p_cache_plus_p_package": (combined, {}),
+ }
+ causal = {}
+ latency = {}
+ with torch.inference_mode():
+ base_logits = wrapper(**inputs, use_cache=False).logits[:, -1].float()
+ for label, (store, condition_kwargs) in conditions.items():
+ output = wrapper(
+ **inputs, p_store=store, query_entity_surfaces=["user"],
+ collect_telemetry=True, use_cache=False, **condition_kwargs,
+ ).logits[:, -1].float()
+ probability = F.softmax(output, dim=-1)
+ route = wrapper.route_telemetry[-1] if wrapper.route_telemetry else None
+ selected_index = None if route is None else int(route.indices[0, -1, 0])
+ causal[label] = {
+ "alice_logit": float(output[0, alice_id]),
+ "bob_logit": float(output[0, bob_id]),
+ "alice_probability": float(probability[0, alice_id]),
+ "bob_probability": float(probability[0, bob_id]),
+ "generated": tokenizer.decode([int(output.argmax(-1))]),
+ "kl_from_base": float(F.kl_div(
+ F.log_softmax(base_logits, dim=-1),
+ F.softmax(output, dim=-1), reduction="batchmean",
+ )),
+ "gate": 0.0 if not wrapper.gate_telemetry else float(torch.stack([
+ gate[:, -1].float().mean() for gate in wrapper.gate_telemetry
+ ]).mean()),
+ "selected_index": selected_index,
+ "selected_state": (
+ None if store is None or selected_index is None
+ else store.cache.labels[selected_index]
+ ),
+ "router_score": (
+ None if route is None else float(route.scores[0, -1, 0])
+ ),
+ "router_accepted": (
+ False if route is None else bool(route.accepted[0, -1])
+ ),
+ "active_state_vram_bytes": _entry_store_bytes(store),
+ }
+ for _ in range(2):
+ wrapper(
+ **inputs, p_store=store, query_entity_surfaces=["user"],
+ use_cache=False, **condition_kwargs,
+ )
+ torch.cuda.synchronize()
+ started = time.perf_counter()
+ for _ in range(5):
+ wrapper(
+ **inputs, p_store=store, query_entity_surfaces=["user"],
+ use_cache=False, **condition_kwargs,
+ )
+ torch.cuda.synchronize()
+ latency[label] = (time.perf_counter() - started) / 5
+
+ rp_inputs = tokenizer(
+ list(NATURAL_RP), return_tensors="pt", padding=True,
+ add_special_tokens=False,
+ ).to("cuda")
+ frozen_rp = wrapper(
+ **rp_inputs, labels=rp_inputs.input_ids, use_cache=False
+ )
+ natural = {}
+ for label, store in (
+ ("frozen_base", None),
+ ("p_cache_only", p_cache),
+ ("p_package_relevant", activations["package_a"].store),
+ ("p_package_irrelevant", activations["irrelevant"].store),
+ ("p_package_contradictory_low_confidence", activations["low_confidence"].store),
+ ("p_cache_plus_p_package", combined),
+ ):
+ output = wrapper(
+ **rp_inputs, labels=rp_inputs.input_ids, p_store=store,
+ use_cache=False
+ )
+ natural[label] = {
+ "loss": float(output.loss),
+ "kl_from_base": float(F.kl_div(
+ F.log_softmax(frozen_rp.logits[:, -1].float(), dim=-1),
+ F.softmax(output.logits[:, -1].float(), dim=-1),
+ reduction="batchmean",
+ )),
+ "samples": [
+ tokenizer.decode([int(row.argmax())])
+ for row in output.logits[:, -1]
+ ],
+ }
+
+ base_gradients = sum(parameter.grad is not None for parameter in base.parameters())
+ activation_paths = {
+ "package_a": package_a,
+ "package_b": package_b,
+ "irrelevant": package_irrelevant,
+ "low_confidence": package_low,
+ "context_technical": package_context,
+ "context_creative": package_context,
+ }
+ active_memory = {
+ label: {
+ "loaded_entries": len(activation.selection.entries),
+ "canonical_store_bytes": _entry_store_bytes(activation.store),
+ "logical_disk_bytes_read": activation.selection.logical_bytes_read,
+ "package_disk_bytes": activation_paths[label].stat().st_size,
+ }
+ for label, activation in activations.items()
+ }
+ result = {
+ "causal": causal,
+ "natural_interaction": natural,
+ "latency_seconds": latency,
+ "active_memory": active_memory,
+ "inactive_package_entries": inactive_entries,
+ "inactive_vram_delta_bytes": after_inactive - before_inactive,
+ "base_parameters_with_grad": base_gradients,
+ "source_tokens_in_recent_kv": 0,
+ "extra_prompt_tokens": 0,
+ "full_package_uploaded_to_cuda": False,
+ "candidate_accuracy": {
+ "package_a_alice": float(causal["p_package_a_relevant"]["alice_logit"] > causal["p_package_a_relevant"]["bob_logit"]),
+ "package_b_bob": float(causal["p_package_b_relevant"]["bob_logit"] > causal["p_package_b_relevant"]["alice_logit"]),
+ "irrelevant_matches_base": float(
+ causal["p_package_irrelevant"]["alice_logit"] == causal["frozen_base"]["alice_logit"]
+ and causal["p_package_irrelevant"]["bob_logit"] == causal["frozen_base"]["bob_logit"]
+ ),
+ "context_technical_alice": float(
+ causal["p_package_context_technical"]["alice_logit"]
+ > causal["p_package_context_technical"]["bob_logit"]
+ ),
+ "context_creative_bob": float(
+ causal["p_package_context_creative"]["bob_logit"]
+ > causal["p_package_context_creative"]["alice_logit"]
+ ),
+ },
+ "relevant_personality_chat": {
+ "target": "Alice",
+ "base_target_loss": float(F.cross_entropy(
+ base_logits, torch.tensor([alice_id], device="cuda")
+ )),
+ "package_target_loss": -math.log(max(
+ causal["p_package_a_relevant"]["alice_probability"], 1e-30
+ )),
+ "target_accuracy": float(
+ causal["p_package_a_relevant"]["alice_logit"]
+ > causal["p_package_a_relevant"]["bob_logit"]
+ ),
+ "generated": causal["p_package_a_relevant"]["generated"],
+ },
+ }
+ for session in sessions.values():
+ session.close()
+ context_session.close()
+ wrapper.close()
+ del wrapper, base, package, canonical_router
+ gc.collect()
+ torch.cuda.empty_cache()
+ return result
+
+
+def run_personality_package_experiment(
+ *,
+ output_package: Path,
+ model_path: Path | None = None,
+ ttl_path: Path | None = None,
+ router_path: Path | None = None,
+ representation: FactorizedStateRepresentation | None = None,
+ growth_counts=(100, 1_000, 10_000, 100_000),
+) -> dict[str, object]:
+ mechanical = build_proof_package(output_package)
+ with tempfile.TemporaryDirectory(prefix="pcm-ppkg-") as temporary:
+ root = Path(temporary)
+ deterministic = deterministic_serialization_proof(root)
+ growth = growth_benchmark(root, counts=growth_counts)
+ cuda = None
+ if model_path and ttl_path and router_path and representation is not None:
+ cuda = cuda_ttl_benchmark(
+ model_path=model_path, ttl_path=ttl_path,
+ router_path=router_path, proof_package_path=output_package,
+ representation=representation, workdir=root,
+ )
+ durability_started = time.perf_counter()
+ with PersonalityPackage(output_package) as restored:
+ durability = {
+ "checksum_valid": True,
+ "active_entries_after_restart": len(restored.entries(
+ status=PersonalityStatus.ACTIVE.value
+ )),
+ "conversation_replay_required": False,
+ "cold_load_seconds": time.perf_counter() - durability_started,
+ }
+ return {
+ "experiment": "phase-b-personality-package-v1",
+ "format": "pcm-personality-package-v1",
+ "protocol": "pcm-canonical-personality-v1",
+ "phase_c_started": False,
+ "lora_used": False,
+ "base_weights_modified": False,
+ "mechanical": mechanical,
+ "deterministic_serialization": deterministic,
+ "growth": growth,
+ "durability": durability,
+ "cuda_ttl": cuda,
+ "package_path": str(output_package),
+ "package_file_sha256": _file_sha256(output_package),
+ }
diff --git a/src/pcm/planner/pythia_split_translate.py b/src/pcm/planner/pythia_split_translate.py
new file mode 100644
index 0000000000000000000000000000000000000000..712a431725e552dfd071fa5afac92e223b37d060
--- /dev/null
+++ b/src/pcm/planner/pythia_split_translate.py
@@ -0,0 +1,201 @@
+"""Frozen-Pythia attachment for the split canonical router/translator."""
+
+from __future__ import annotations
+
+import torch
+from torch import Tensor, nn
+from pathlib import Path
+
+from pcm.planner.canonical import CanonicalPStore
+from pcm.planner.split_translator import (
+ ByteEntityEncoder,
+ CanonicalPRouter,
+ CanonicalRouterIndex,
+ RouteResult,
+ FactorizedCanonicalQuery,
+ SplitPTranslatePackage,
+ config_checksum,
+)
+
+
+def pythia_model_identifier(base_model: nn.Module) -> str:
+ configured = str(getattr(base_model.config, "_name_or_path", "")).strip()
+ if configured and Path(configured).is_absolute():
+ configured = Path(configured).name
+ return configured or str(getattr(base_model.config, "model_type", "gpt_neox"))
+
+
+class PythiaSplitTranslatedModel(nn.Module):
+ def __init__(
+ self,
+ base_model: nn.Module,
+ package: SplitPTranslatePackage,
+ router: CanonicalPRouter,
+ entity_encoder: ByteEntityEncoder,
+ ) -> None:
+ super().__init__()
+ if not hasattr(base_model, "gpt_neox"):
+ raise TypeError("base model must expose GPT-NeoX transformer layers")
+ package.validate_compatibility(
+ model_id=pythia_model_identifier(base_model),
+ model_hidden_width=int(base_model.config.hidden_size),
+ attachment_layers=package.config.attachment_layers,
+ model_config_sha256=config_checksum(base_model.config),
+ )
+ self.base_model = base_model
+ self.package = package
+ self.router = router
+ self.entity_encoder = entity_encoder
+ for parameter in base_model.parameters():
+ parameter.requires_grad_(False)
+ layers = base_model.gpt_neox.layers
+ if any(index < 0 or index >= len(layers) for index in package.config.attachment_layers):
+ raise IndexError("split translator attachment layer is outside Pythia depth")
+ self._store: CanonicalPStore | None = None
+ self._index: CanonicalRouterIndex | None = None
+ self._oracle_indices: Tensor | None = None
+ self._query_entity_anchor: Tensor | None = None
+ self._gate_enabled = True
+ self._injection_enabled = True
+ self._collect = False
+ self._gate_telemetry: list[Tensor] = []
+ self._route_telemetry: list[RouteResult] = []
+ self._query_telemetry: list[FactorizedCanonicalQuery] = []
+ self._handles = [
+ layers[index].register_forward_hook(self._hook)
+ for index in package.config.attachment_layers
+ ]
+ self.base_model.eval()
+
+ def _oracle_route(self, query, hidden: Tensor) -> RouteResult:
+ assert self._index is not None and self._oracle_indices is not None
+ scores, features = self.router.all_scores(query, self._index)
+ batch, sequence = hidden.shape[:2]
+ indices = self._oracle_indices.to(hidden.device).view(batch, 1, 1).expand(batch, sequence, 1)
+ selected_scores = scores.gather(-1, indices)
+ selected_features = features.gather(
+ -2, indices.unsqueeze(-1).expand(batch, sequence, 1, 4)
+ )
+ return RouteResult(
+ indices=indices,
+ scores=selected_scores,
+ weights=torch.ones_like(selected_scores),
+ features=selected_features,
+ accepted=torch.ones((batch, sequence), dtype=torch.bool, device=hidden.device),
+ has_valid=True,
+ )
+
+ def _hook(self, _module, _inputs, hidden: Tensor):
+ if self._store is None or self._store.cache.occupied == 0:
+ return hidden
+ assert self._index is not None
+ entity_anchor = None
+ if self._query_entity_anchor is not None:
+ entity_anchor = self._query_entity_anchor[:, None, :].expand(
+ hidden.shape[0], hidden.shape[1], -1
+ )
+ query = self.package.query_projector(hidden, entity_anchor=entity_anchor)
+ if self._collect:
+ self._query_telemetry.append(FactorizedCanonicalQuery(
+ entity=query.entity.detach(),
+ relation_logits=query.relation_logits.detach(),
+ metadata_logits=query.metadata_logits.detach(),
+ ))
+ route = (
+ self._oracle_route(query, hidden)
+ if self._oracle_indices is not None
+ else self.router.route(query, self._index, top_k=self.package.config.top_k)
+ )
+ if self._collect:
+ self._route_telemetry.append(RouteResult(
+ indices=route.indices.detach(), scores=route.scores.detach(),
+ weights=route.weights.detach(), features=route.features.detach(),
+ accepted=route.accepted.detach(),
+ has_valid=route.has_valid,
+ ))
+ if not self._injection_enabled or not route.has_valid:
+ return hidden
+ canonical = self._store.canonical_values.to(
+ device=hidden.device, dtype=route.weights.dtype
+ )
+ selected = canonical[route.indices]
+ pooled = torch.einsum("...k,...kd->...d", route.weights, selected)
+ translated = self.package.value_translator(pooled)
+ route_features = torch.einsum(
+ "...k,...kf->...f", route.weights, route.features
+ )
+ gate = (
+ self.package.gate(hidden, translated, route_features)
+ if self._gate_enabled
+ else torch.ones(hidden.shape[:-1], device=hidden.device, dtype=translated.dtype)
+ )
+ gate = gate * route.accepted.to(gate.dtype)
+ if self._collect:
+ self._gate_telemetry.append(gate.detach())
+ return hidden + (gate.unsqueeze(-1) * translated).to(hidden.dtype)
+
+ def train(self, mode: bool = True):
+ super().train(mode)
+ self.base_model.eval()
+ self.package.train(mode)
+ self.router.train(mode)
+ return self
+
+ def forward(
+ self,
+ *args,
+ p_store: CanonicalPStore | None = None,
+ query_entity_surfaces: list[str] | tuple[str, ...] | None = None,
+ oracle_indices: Tensor | None = None,
+ gate_enabled: bool = True,
+ injection_enabled: bool = True,
+ collect_telemetry: bool = False,
+ **kwargs,
+ ):
+ if self._store is not None:
+ raise RuntimeError("PythiaSplitTranslatedModel is not reentrant")
+ self._store = p_store
+ self._oracle_indices = oracle_indices
+ if query_entity_surfaces is not None:
+ if "input_ids" in kwargs and len(query_entity_surfaces) != kwargs["input_ids"].shape[0]:
+ raise ValueError("query entity surface count must match the input batch")
+ self._query_entity_anchor = self.entity_encoder(
+ list(query_entity_surfaces)
+ ).to(next(self.package.parameters()).device)
+ else:
+ self._query_entity_anchor = None
+ self._gate_enabled = gate_enabled
+ self._injection_enabled = injection_enabled
+ self._collect = collect_telemetry
+ self._gate_telemetry.clear()
+ self._route_telemetry.clear()
+ self._query_telemetry.clear()
+ if p_store is not None and p_store.cache.occupied:
+ self._index = self.router.build_index(
+ p_store, self.entity_encoder, device=next(self.package.parameters()).device
+ )
+ try:
+ return self.base_model(*args, **kwargs)
+ finally:
+ self._store = None
+ self._index = None
+ self._oracle_indices = None
+ self._query_entity_anchor = None
+ self._collect = False
+
+ @property
+ def gate_telemetry(self):
+ return tuple(self._gate_telemetry)
+
+ @property
+ def route_telemetry(self):
+ return tuple(self._route_telemetry)
+
+ @property
+ def query_telemetry(self):
+ return tuple(self._query_telemetry)
+
+ def close(self):
+ for handle in self._handles:
+ handle.remove()
+ self._handles.clear()
diff --git a/src/pcm/planner/representation.py b/src/pcm/planner/representation.py
new file mode 100644
index 0000000000000000000000000000000000000000..ff1fd1b1cceadd0941fc104769eebacb1894bdc5
--- /dev/null
+++ b/src/pcm/planner/representation.py
@@ -0,0 +1,240 @@
+"""Factorized, contrastively trained canonical state representation probes."""
+
+from __future__ import annotations
+
+from dataclasses import dataclass
+import random
+
+import torch
+from torch import Tensor, nn
+import torch.nn.functional as F
+
+
+CANONICAL = 0
+CONTRADICTED = 1
+HISTORICAL = 2
+INFERRED = 3
+
+
+class FactorizedStateRepresentation(nn.Module):
+ """Separate entity/relation/value/metadata fields projected to a P slot."""
+
+ def __init__(
+ self,
+ entities: int,
+ relations: int,
+ values: int,
+ metadata: int = 4,
+ *,
+ field_width: int = 128,
+ slot_width: int = 512,
+ ) -> None:
+ super().__init__()
+ self.entity = nn.Embedding(entities, field_width)
+ self.relation = nn.Embedding(relations, field_width)
+ self.value = nn.Embedding(values, field_width)
+ self.metadata = nn.Embedding(metadata, field_width)
+ factor_width = field_width * 4
+ self.slot_projection = nn.Linear(factor_width, slot_width, bias=False)
+ self.tuple_projection = nn.Linear(factor_width, slot_width, bias=False)
+ self.query_projection = nn.Linear(field_width * 3, slot_width, bias=False)
+ self.entity_decoder = nn.Linear(slot_width, entities)
+ self.relation_decoder = nn.Linear(slot_width, relations)
+ self.value_decoder = nn.Linear(slot_width, values)
+ self.metadata_decoder = nn.Linear(slot_width, metadata)
+ self.temperature = nn.Parameter(torch.tensor(0.07))
+
+ def factors(self, entity: Tensor, relation: Tensor, value: Tensor, metadata: Tensor) -> Tensor:
+ return torch.cat((
+ self.entity(entity),
+ self.relation(relation),
+ self.value(value),
+ self.metadata(metadata),
+ ), dim=-1)
+
+ def encode(self, entity: Tensor, relation: Tensor, value: Tensor, metadata: Tensor) -> Tensor:
+ return F.normalize(self.slot_projection(self.factors(entity, relation, value, metadata)), dim=-1)
+
+ def tuple_anchor(self, entity: Tensor, relation: Tensor, value: Tensor, metadata: Tensor) -> Tensor:
+ return F.normalize(self.tuple_projection(self.factors(entity, relation, value, metadata)), dim=-1)
+
+ def query(self, entity: Tensor, relation: Tensor, metadata: Tensor) -> Tensor:
+ fields = torch.cat((self.entity(entity), self.relation(relation), self.metadata(metadata)), dim=-1)
+ return F.normalize(self.query_projection(fields), dim=-1)
+
+ def scores(self, query: Tensor, slots: Tensor) -> Tensor:
+ temperature = self.temperature.clamp(0.02, 1.0)
+ return torch.einsum("bd,bkd->bk", query, slots) / temperature
+
+ def decode(self, slots: Tensor) -> tuple[Tensor, Tensor, Tensor, Tensor]:
+ return (
+ self.entity_decoder(slots),
+ self.relation_decoder(slots),
+ self.value_decoder(slots),
+ self.metadata_decoder(slots),
+ )
+
+
+@dataclass(frozen=True)
+class CompositionConfig:
+ entities: int = 24
+ relations: int = 3
+ values: int = 36
+ candidates: int = 4
+
+
+def is_held_out(entity: int, relation: int, value: int) -> bool:
+ return (entity * 31 + relation * 17 + value * 13) % 5 == 0
+
+
+def composition_splits(config: CompositionConfig):
+ train, held_out = [], []
+ for entity in range(config.entities):
+ for relation in range(config.relations):
+ for value in range(config.values):
+ target = held_out if is_held_out(entity, relation, value) else train
+ target.append((entity, relation, value))
+ return train, held_out
+
+
+def _same_split_alternative(entity, relation, value, size, want_held_out, field):
+ for offset in range(1, size):
+ if field == "value":
+ candidate = (entity, relation, (value + offset) % size)
+ else:
+ candidate = ((entity + offset) % size, relation, value)
+ if is_held_out(*candidate) == want_held_out:
+ return candidate
+ raise RuntimeError("unable to construct composition-preserving negative")
+
+
+def candidate_tuples(positive, config: CompositionConfig, *, held_out: bool):
+ entity, relation, value = positive
+ wrong_value = _same_split_alternative(
+ entity, relation, value, config.values, held_out, "value"
+ )
+ wrong_entity = _same_split_alternative(
+ entity, relation, value, config.entities, held_out, "entity"
+ )
+ return (
+ (entity, relation, value, CANONICAL, "correct"),
+ (*wrong_value, CONTRADICTED, "wrong_value"),
+ (*wrong_entity, CANONICAL, "wrong_entity"),
+ (entity, relation, value, HISTORICAL, "historical"),
+ )
+
+
+def make_batch(combinations, config, batch_size, rng, *, held_out, permute=True):
+ selected = [combinations[rng.randrange(len(combinations))] for _ in range(batch_size)]
+ candidates, targets, kinds = [], [], []
+ for positive in selected:
+ rows = list(candidate_tuples(positive, config, held_out=held_out))
+ if permute:
+ rng.shuffle(rows)
+ candidates.append([row[:4] for row in rows])
+ kinds.append([row[4] for row in rows])
+ targets.append(next(index for index, row in enumerate(rows) if row[4] == "correct"))
+ positive = torch.tensor(selected, dtype=torch.long)
+ return positive, torch.tensor(candidates, dtype=torch.long), torch.tensor(targets), kinds
+
+
+def representation_loss(model, positive, candidates, targets):
+ entity, relation, value = positive.T
+ flat = candidates.view(-1, 4)
+ slots = model.encode(*flat.T).view(candidates.shape[0], candidates.shape[1], -1)
+ query = model.query(entity, relation, torch.full_like(entity, CANONICAL))
+ retrieval = F.cross_entropy(model.scores(query, slots), targets)
+ anchor = model.tuple_anchor(entity, relation, value, torch.full_like(entity, CANONICAL))
+ contrastive = F.cross_entropy(model.scores(anchor, slots), targets)
+ decoded = model.decode(slots)
+ canonical = sum(
+ F.cross_entropy(logits.flatten(0, 1), flat[:, field])
+ for field, logits in enumerate(decoded)
+ )
+ return retrieval + contrastive + 0.5 * canonical
+
+
+def evaluate_representation(model, combinations, config, *, permutations=8, seed=101):
+ model.eval()
+ totals = {"correct": 0, "wrong_value": 0, "wrong_entity": 0, "historical": 0}
+ count = 0
+ decoded = torch.zeros(4)
+ stable = 0
+ rng = random.Random(seed)
+ with torch.inference_mode():
+ for positive in combinations:
+ chosen_values = []
+ for _ in range(permutations):
+ pos, candidates, target, kinds = make_batch(
+ [positive], config, 1, rng, held_out=True, permute=True
+ )
+ flat = candidates.view(-1, 4)
+ slots = model.encode(*flat.T).view(1, config.candidates, -1)
+ query = model.query(pos[:, 0], pos[:, 1], torch.zeros(1, dtype=torch.long))
+ scores = model.scores(query, slots)[0]
+ correct_index = int(target[0])
+ prediction = int(scores.argmax())
+ selected_slot = slots[0, prediction]
+ selected_value = int(model.value_decoder(selected_slot).argmax())
+ chosen_values.append(selected_value)
+ if prediction == correct_index and selected_value == positive[2]:
+ totals["correct"] += 1
+ for index, kind in enumerate(kinds[0]):
+ if kind != "correct" and scores[correct_index] > scores[index]:
+ totals[kind] += 1
+ decoded_logits = model.decode(slots[0, correct_index])
+ truth = candidates[0, correct_index]
+ decoded += torch.tensor([
+ int(logits.argmax() == truth[field])
+ for field, logits in enumerate(decoded_logits)
+ ])
+ count += 1
+ stable += int(len(set(chosen_values)) == 1 and chosen_values[0] == positive[2])
+ return {
+ "p_only_state_recovery": totals["correct"] / count,
+ "hard_negative_accuracy": {
+ kind: totals[kind] / count for kind in ("wrong_value", "wrong_entity", "historical")
+ },
+ "canonical_decode_accuracy": {
+ field: float(decoded[index] / count)
+ for index, field in enumerate(("entity", "relation", "value", "metadata"))
+ },
+ "permutation_stability": stable / len(combinations),
+ "held_out_combinations": len(combinations),
+ "permutations_per_combination": permutations,
+ }
+
+
+def train_and_probe_representation(
+ *, steps=600, batch_size=64, slot_width=512, seed=97, evaluation_limit=256
+):
+ torch.manual_seed(seed)
+ config = CompositionConfig()
+ train, held_out = composition_splits(config)
+ model = FactorizedStateRepresentation(
+ config.entities, config.relations, config.values, slot_width=slot_width
+ )
+ optimizer = torch.optim.AdamW(model.parameters(), lr=3e-3)
+ rng = random.Random(seed)
+ losses = []
+ model.train()
+ for _ in range(steps):
+ positive, candidates, targets, _ = make_batch(
+ train, config, batch_size, rng, held_out=False
+ )
+ loss = representation_loss(model, positive, candidates, targets)
+ optimizer.zero_grad(set_to_none=True)
+ loss.backward()
+ optimizer.step()
+ losses.append(float(loss.detach()))
+ probe = evaluate_representation(
+ model, held_out[:evaluation_limit], config, permutations=8, seed=seed + 1
+ )
+ probe.update({
+ "training_loss_first": losses[0],
+ "training_loss_last": losses[-1],
+ "train_combinations": len(train),
+ "total_held_out_combinations": len(held_out),
+ "slot_width": slot_width,
+ })
+ return model, config, probe
diff --git a/src/pcm/planner/split_translator.py b/src/pcm/planner/split_translator.py
new file mode 100644
index 0000000000000000000000000000000000000000..b6199686f880e4d9b815a641913d551d02beeb1a
--- /dev/null
+++ b/src/pcm/planner/split_translator.py
@@ -0,0 +1,404 @@
+"""Split model-to-canonical routing and canonical-to-model translation."""
+
+from __future__ import annotations
+
+from dataclasses import asdict, dataclass
+import json
+from pathlib import Path
+
+import torch
+from torch import Tensor, nn
+import torch.nn.functional as F
+from safetensors import safe_open
+from safetensors.torch import load_file, save_file
+
+from pcm.planner.canonical import CanonicalPStore
+from pcm.planner.canonical import (
+ CANONICAL_P_PROTOCOL,
+ model_config_checksum,
+ tensor_state_checksum,
+)
+from pcm.planner.cache import Freshness
+
+
+SPLIT_TRANSLATE_FORMAT = "pcm-split-translate-v1"
+CANONICAL_ROUTER_FORMAT = "pcm-canonical-router-v1"
+
+
+def tensor_checksum(state: dict[str, Tensor]) -> str:
+ return tensor_state_checksum(state)
+
+
+class ByteEntityEncoder(nn.Module):
+ """Tokenizer-independent signed byte n-gram features for open entity names."""
+
+ def __init__(self, width: int = 128) -> None:
+ super().__init__()
+ if width < 32:
+ raise ValueError("byte entity width must be at least 32")
+ self.width = width
+
+ def encode_one(self, surface: str) -> Tensor:
+ data = surface.strip().casefold().encode("utf-8")
+ if not data:
+ raise ValueError("entity surface form cannot be empty")
+ vector = torch.zeros(self.width, dtype=torch.float32)
+ for position, byte in enumerate(data):
+ vector[(byte * 17 + position * 31) % self.width] += 1.0
+ for position, (left, right) in enumerate(zip(data, data[1:])):
+ bucket = (left * 257 + right * 17 + position * 13) % self.width
+ sign = 1.0 if ((left + right + position) & 1) == 0 else -1.0
+ vector[bucket] += 0.5 * sign
+ return F.normalize(vector, dim=0)
+
+ def forward(self, surfaces: list[str] | tuple[str, ...]) -> Tensor:
+ return torch.stack([self.encode_one(surface) for surface in surfaces])
+
+
+@dataclass
+class FactorizedCanonicalQuery:
+ entity: Tensor
+ relation_logits: Tensor
+ metadata_logits: Tensor
+
+
+class ModelToCanonicalQueryProjector(nn.Module):
+ def __init__(
+ self,
+ model_hidden_width: int,
+ *,
+ entity_width: int = 128,
+ relation_count: int = 3,
+ metadata_count: int = 4,
+ ) -> None:
+ super().__init__()
+ self.model_hidden_width = model_hidden_width
+ self.entity_width = entity_width
+ self.relation_count = relation_count
+ self.metadata_count = metadata_count
+ self.norm = nn.LayerNorm(model_hidden_width)
+ self.shared = nn.Linear(model_hidden_width, 512)
+ self.entity_head = nn.Linear(512, entity_width)
+ self.relation_head = nn.Linear(512, relation_count)
+ self.metadata_head = nn.Linear(512, metadata_count)
+
+ def forward(self, hidden: Tensor, entity_anchor: Tensor | None = None) -> FactorizedCanonicalQuery:
+ hidden = hidden.detach().to(self.shared.weight.dtype)
+ features = F.gelu(self.shared(self.norm(hidden)))
+ predicted_entity = F.normalize(self.entity_head(features), dim=-1)
+ if entity_anchor is not None:
+ predicted_entity = F.normalize(
+ entity_anchor.to(device=hidden.device, dtype=predicted_entity.dtype), dim=-1
+ )
+ return FactorizedCanonicalQuery(
+ entity=predicted_entity,
+ relation_logits=self.relation_head(features),
+ metadata_logits=self.metadata_head(features),
+ )
+
+
+class FrozenLexicalAnchorProjector(nn.Module):
+ """Experimental model-native lexical anchor mapped into canonical entity space."""
+
+ def __init__(self, model_hidden_width: int, entity_width: int = 128) -> None:
+ super().__init__()
+ self.norm = nn.LayerNorm(model_hidden_width)
+ self.projection = nn.Sequential(
+ nn.Linear(model_hidden_width, 256), nn.GELU(), nn.Linear(256, entity_width)
+ )
+
+ def forward(self, lexical_hidden: Tensor) -> Tensor:
+ return F.normalize(self.projection(self.norm(lexical_hidden.detach().float())), dim=-1)
+
+
+@dataclass(frozen=True)
+class RouterConfig:
+ entity_width: int = 128
+ relation_count: int = 3
+ metadata_count: int = 4
+ format: str = CANONICAL_ROUTER_FORMAT
+ canonical_protocol: str = CANONICAL_P_PROTOCOL
+ architecture: str = "canonical_factor_router_v1"
+
+
+@dataclass
+class CanonicalRouterIndex:
+ entity: Tensor
+ relation_id: Tensor
+ metadata_id: Tensor
+ valid: Tensor
+
+
+@dataclass
+class RouteResult:
+ indices: Tensor
+ scores: Tensor
+ weights: Tensor
+ features: Tensor
+ accepted: Tensor
+ has_valid: bool
+
+
+class CanonicalPRouter(nn.Module):
+ """Universal canonical-only scorer; it has no model-hidden dimensions."""
+
+ def __init__(self, config: RouterConfig = RouterConfig()) -> None:
+ super().__init__()
+ self.config = config
+ self.scorer = nn.Linear(4, 1)
+ self.register_buffer("acceptance_threshold", torch.tensor(0.0))
+ with torch.no_grad():
+ self.scorer.weight.copy_(torch.tensor([[8.0, 4.0, 2.0, 2.0]]))
+ self.scorer.bias.zero_()
+
+ def build_index(
+ self,
+ store: CanonicalPStore,
+ encoder: ByteEntityEncoder,
+ *,
+ device: str | torch.device,
+ ) -> CanonicalRouterIndex:
+ surfaces = []
+ for valid, label in zip(store.valid.tolist(), store.cache.labels):
+ if valid and not label:
+ raise ValueError("routable canonical P slots require an entity surface label")
+ surfaces.append(label if label else "")
+ entity = encoder(surfaces).to(device)
+ return CanonicalRouterIndex(
+ entity=entity,
+ relation_id=store.relation_id.to(device),
+ metadata_id=store.canonical_metadata_id.to(device),
+ valid=(
+ store.valid
+ & (store.cache.freshness != int(Freshness.STALE))
+ ).to(device),
+ )
+
+ def all_scores(
+ self, query: FactorizedCanonicalQuery, index: CanonicalRouterIndex
+ ) -> tuple[Tensor, Tensor]:
+ entity = torch.einsum("...d,sd->...s", query.entity.float(), index.entity.float())
+ relation_probability = F.softmax(query.relation_logits.float(), dim=-1)
+ relation_ids = index.relation_id.clamp_min(0)
+ relation = relation_probability[..., relation_ids]
+ metadata_probability = F.softmax(query.metadata_logits.float(), dim=-1)
+ metadata_ids = index.metadata_id.clamp_min(0)
+ metadata = metadata_probability[..., metadata_ids]
+ current = (index.metadata_id == 0).float().view(
+ *((1,) * (entity.ndim - 1)), -1
+ ).expand_as(entity)
+ features = torch.stack((entity, relation, metadata, current), dim=-1)
+ scores = self.scorer(features).squeeze(-1)
+ valid = index.valid.view(*((1,) * (scores.ndim - 1)), -1)
+ return scores.masked_fill(~valid, -torch.inf), features
+
+ def route(
+ self,
+ query: FactorizedCanonicalQuery,
+ index: CanonicalRouterIndex,
+ *,
+ top_k: int = 1,
+ ) -> RouteResult:
+ if top_k <= 0:
+ raise ValueError("top_k must be positive")
+ scores, features = self.all_scores(query, index)
+ count = min(top_k, scores.shape[-1])
+ if not bool(index.valid.any()):
+ shape = (*scores.shape[:-1], count)
+ return RouteResult(
+ indices=torch.zeros(shape, dtype=torch.long, device=scores.device),
+ scores=torch.full(shape, -torch.inf, device=scores.device),
+ weights=torch.zeros(shape, device=scores.device),
+ features=torch.zeros((*shape, 4), device=scores.device),
+ accepted=torch.zeros(scores.shape[:-1], dtype=torch.bool, device=scores.device),
+ has_valid=False,
+ )
+ selected_scores, indices = scores.topk(count, dim=-1)
+ weights = torch.softmax(selected_scores, dim=-1)
+ selected_features = features.gather(
+ -2, indices.unsqueeze(-1).expand(*indices.shape, features.shape[-1])
+ )
+ return RouteResult(
+ indices=indices,
+ scores=selected_scores,
+ weights=weights,
+ features=selected_features,
+ accepted=selected_scores[..., 0] >= self.acceptance_threshold,
+ has_valid=bool(index.valid.any()),
+ )
+
+ def calibrate_acceptance(self, positive_scores: Tensor, negative_scores: Tensor) -> float:
+ positive_scores = positive_scores.detach().float().flatten()
+ negative_scores = negative_scores.detach().float().flatten()
+ candidates = torch.unique(torch.cat((positive_scores, negative_scores))).sort().values
+ if candidates.numel() > 1:
+ candidates = (candidates[:-1] + candidates[1:]) / 2
+ best_threshold = candidates[0]
+ best_balanced = -1.0
+ for threshold in candidates:
+ true_positive = (positive_scores >= threshold).float().mean()
+ true_negative = (negative_scores < threshold).float().mean()
+ balanced = float((true_positive + true_negative) / 2)
+ if balanced > best_balanced:
+ best_balanced = balanced
+ best_threshold = threshold
+ self.acceptance_threshold.copy_(best_threshold.to(self.acceptance_threshold.device))
+ return float(best_balanced)
+
+ def save(self, path: str | Path) -> None:
+ state = {name: value.detach().cpu() for name, value in self.state_dict().items()}
+ save_file(state, str(path), metadata={
+ "format": CANONICAL_ROUTER_FORMAT,
+ "config": json.dumps(asdict(self.config), sort_keys=True),
+ "weights_sha256": tensor_checksum(state),
+ })
+
+ @classmethod
+ def load(cls, path: str | Path, *, device="cpu") -> "CanonicalPRouter":
+ with safe_open(str(path), framework="pt", device="cpu") as handle:
+ metadata = handle.metadata()
+ if metadata.get("format") != CANONICAL_ROUTER_FORMAT:
+ raise ValueError("unsupported canonical router file")
+ config = RouterConfig(**json.loads(metadata["config"]))
+ result = cls(config).to(device)
+ state = load_file(str(path), device=str(device))
+ if tensor_checksum(state) != metadata.get("weights_sha256"):
+ raise ValueError("canonical router checksum does not match")
+ result.load_state_dict(state)
+ return result
+
+
+class CanonicalValueTranslator(nn.Module):
+ def __init__(self, canonical_width: int, model_hidden_width: int) -> None:
+ super().__init__()
+ self.norm = nn.LayerNorm(canonical_width)
+ self.input = nn.Linear(canonical_width, 512)
+ self.output = nn.Linear(512, model_hidden_width)
+
+ def forward(self, canonical: Tensor) -> Tensor:
+ canonical = canonical.to(self.input.weight.dtype)
+ return self.output(F.gelu(self.input(self.norm(canonical))))
+
+
+class SplitInjectionGate(nn.Module):
+ def __init__(self, model_hidden_width: int) -> None:
+ super().__init__()
+ self.hidden_norm = nn.LayerNorm(model_hidden_width)
+ self.value_norm = nn.LayerNorm(model_hidden_width)
+ self.joint = nn.Linear(model_hidden_width * 2 + 4, 64)
+ self.output = nn.Linear(64, 1)
+ nn.init.zeros_(self.output.weight)
+ nn.init.constant_(self.output.bias, -4.0)
+
+ def logits(self, hidden: Tensor, translated: Tensor, route_features: Tensor) -> Tensor:
+ dtype = self.joint.weight.dtype
+ joint = torch.cat((
+ self.hidden_norm(hidden.detach().to(dtype)),
+ self.value_norm(translated.to(dtype)),
+ route_features.to(dtype),
+ ), dim=-1)
+ return self.output(F.gelu(self.joint(joint))).squeeze(-1)
+
+ def forward(self, hidden: Tensor, translated: Tensor, route_features: Tensor) -> Tensor:
+ return torch.sigmoid(self.logits(hidden, translated, route_features))
+
+
+@dataclass(frozen=True)
+class SplitTranslateConfig:
+ model_id: str
+ model_hidden_width: int
+ attachment_layers: tuple[int, ...]
+ canonical_width: int = 512
+ entity_width: int = 128
+ relation_count: int = 3
+ metadata_count: int = 4
+ canonical_protocol: str = CANONICAL_P_PROTOCOL
+ format: str = SPLIT_TRANSLATE_FORMAT
+ architecture: str = "split_query_value_joint_gate_v1"
+ model_revision: str = "local"
+ model_config_sha256: str = "unspecified"
+ top_k: int = 1
+
+ def __post_init__(self):
+ if self.format != SPLIT_TRANSLATE_FORMAT:
+ raise ValueError("unsupported split translator format")
+ if self.canonical_protocol != CANONICAL_P_PROTOCOL:
+ raise ValueError("unsupported canonical P protocol")
+ if self.model_hidden_width <= 0 or self.canonical_width <= 0:
+ raise ValueError("translator widths must be positive")
+ if not self.attachment_layers:
+ raise ValueError("attachment layers cannot be empty")
+ if self.top_k <= 0:
+ raise ValueError("top_k must be positive")
+
+
+class SplitPTranslatePackage(nn.Module):
+ """Model-specific query/value/gate modules; universal router is separate."""
+
+ def __init__(self, config: SplitTranslateConfig) -> None:
+ super().__init__()
+ self.config = config
+ self.query_projector = ModelToCanonicalQueryProjector(
+ config.model_hidden_width,
+ entity_width=config.entity_width,
+ relation_count=config.relation_count,
+ metadata_count=config.metadata_count,
+ )
+ self.value_translator = CanonicalValueTranslator(
+ config.canonical_width, config.model_hidden_width
+ )
+ self.gate = SplitInjectionGate(config.model_hidden_width)
+
+ def validate_compatibility(
+ self,
+ *,
+ model_id: str,
+ model_hidden_width: int,
+ canonical_protocol: str = CANONICAL_P_PROTOCOL,
+ attachment_layers: tuple[int, ...] | None = None,
+ model_config_sha256: str | None = None,
+ ) -> None:
+ errors = []
+ if model_id != self.config.model_id:
+ errors.append("model identifier")
+ if model_hidden_width != self.config.model_hidden_width:
+ errors.append("model hidden width")
+ if canonical_protocol != self.config.canonical_protocol:
+ errors.append("canonical protocol")
+ if attachment_layers is not None and tuple(attachment_layers) != self.config.attachment_layers:
+ errors.append("attachment layers")
+ if (
+ model_config_sha256 is not None
+ and self.config.model_config_sha256 != "unspecified"
+ and model_config_sha256 != self.config.model_config_sha256
+ ):
+ errors.append("model config checksum")
+ if errors:
+ raise ValueError("incompatible split translator: " + ", ".join(errors))
+
+ def save(self, path: str | Path) -> None:
+ state = {name: value.detach().cpu() for name, value in self.state_dict().items()}
+ save_file(state, str(path), metadata={
+ "format": SPLIT_TRANSLATE_FORMAT,
+ "config": json.dumps(asdict(self.config), sort_keys=True),
+ "weights_sha256": tensor_checksum(state),
+ })
+
+ @classmethod
+ def load(cls, path: str | Path, *, device="cpu", dtype=torch.float32):
+ with safe_open(str(path), framework="pt", device="cpu") as handle:
+ metadata = handle.metadata()
+ if metadata.get("format") != SPLIT_TRANSLATE_FORMAT:
+ raise ValueError("unsupported split translator file")
+ raw = json.loads(metadata["config"])
+ raw["attachment_layers"] = tuple(raw["attachment_layers"])
+ result = cls(SplitTranslateConfig(**raw)).to(device=device, dtype=dtype)
+ state = load_file(str(path), device=str(device))
+ if tensor_checksum(state) != metadata.get("weights_sha256"):
+ raise ValueError("split translator checksum does not match")
+ result.load_state_dict({name: value.to(dtype=dtype) for name, value in state.items()})
+ return result
+
+
+def config_checksum(config: object) -> str:
+ return model_config_checksum(config)
diff --git a/src/pcm/planner/split_translator_eval.py b/src/pcm/planner/split_translator_eval.py
new file mode 100644
index 0000000000000000000000000000000000000000..73158205b7feb29b4f93e536666a3784af9ade94
--- /dev/null
+++ b/src/pcm/planner/split_translator_eval.py
@@ -0,0 +1,971 @@
+"""Exact staged evaluation of split canonical routing and value translation."""
+
+from __future__ import annotations
+
+import gc
+from pathlib import Path
+import random
+import time
+
+import torch
+import torch.nn.functional as F
+from transformers import AutoModelForCausalLM, AutoTokenizer
+
+from pcm.planner.canonical import CanonicalPConfig, CanonicalPStore
+from pcm.planner.cache import SlotSource
+from pcm.planner.pythia_split_translate import PythiaSplitTranslatedModel
+from pcm.planner.pythia_split_translate import pythia_model_identifier
+from pcm.planner.representation import CANONICAL, HISTORICAL, FactorizedStateRepresentation
+from pcm.planner.split_translator import (
+ ByteEntityEncoder,
+ CanonicalPRouter,
+ CanonicalRouterIndex,
+ FactorizedCanonicalQuery,
+ FrozenLexicalAnchorProjector,
+ RouterConfig,
+ SplitPTranslatePackage,
+ SplitTranslateConfig,
+ config_checksum,
+)
+from pcm.planner.canonical import CANONICAL_VALUE_LABELS as VALUE_LABELS
+
+
+ADJECTIVES = tuple(
+ "silver gold crimson azure ivory ebony amber jade copper iron crystal shadow bright "
+ "quiet ancient hidden broken little grand northern southern eastern western moon sun "
+ "star river storm winter summer autumn".split()
+)
+NOUNS = tuple(
+ "key ring blade crown lantern compass chalice mirror scroll seal pendant coin map book "
+ "box door tower bridge garden harbor temple forest castle chamber wagon banner stone "
+ "cloak staff mask bell".split()
+)
+RELATION_PROMPTS = (
+ "The owner of the {entity} is",
+ "The current location of the {entity} is",
+ "The current status of the {entity} is",
+)
+HELDOUT_LEADS = (
+ "After a long unrelated scene at the inn, ",
+ "Following several jokes and descriptions of the rainy road, ",
+)
+TRAIN_LEADS = (
+ "After unrelated conversation, ",
+ "With the source state absent from recent context, ",
+)
+SLOT_SIZES = (4, 20, 64, 128, 256, 512)
+
+
+def entity_split():
+ train, heldout = [], []
+ for adjective_index, adjective in enumerate(ADJECTIVES):
+ for noun_index, noun in enumerate(NOUNS):
+ surface = f"{adjective} {noun}"
+ target = heldout if (adjective_index * 31 + noun_index * 17) % 5 == 0 else train
+ target.append(surface)
+ for required in ("silver key", "gold key"):
+ if required in train:
+ train.remove(required)
+ heldout.append(required)
+ return train, heldout
+
+
+def _bytes(parameters) -> int:
+ return sum(parameter.numel() * parameter.element_size() for parameter in parameters)
+
+
+def run_split_translator_experiment(
+ path: str | Path,
+ representation: FactorizedStateRepresentation,
+ *,
+ attachment_count: int,
+ query_steps: int = 400,
+ router_steps: int = 400,
+ value_steps: int = 400,
+ causal_steps: int = 256,
+ seed: int = 307,
+ package_path: str | Path | None = None,
+ router_path: str | Path | None = None,
+):
+ if attachment_count not in (1, 2, 4):
+ raise ValueError("attachment_count must be 1, 2, or 4")
+ if not torch.cuda.is_available():
+ raise RuntimeError("CUDA is required")
+ torch.manual_seed(seed)
+ rng = random.Random(seed)
+ path = Path(path)
+ tokenizer = AutoTokenizer.from_pretrained(path, local_files_only=True)
+ tokenizer.pad_token = tokenizer.eos_token
+ tokenizer.padding_side = "left"
+ base = AutoModelForCausalLM.from_pretrained(
+ path, local_files_only=True, dtype=torch.float16, low_cpu_mem_usage=True
+ ).to("cuda").eval()
+ depth = len(base.gpt_neox.layers)
+ layers = tuple(range(depth - attachment_count, depth))
+ package = SplitPTranslatePackage(SplitTranslateConfig(
+ model_id=pythia_model_identifier(base),
+ model_hidden_width=int(base.config.hidden_size),
+ attachment_layers=layers,
+ model_config_sha256=config_checksum(base.config),
+ top_k=1,
+ )).to("cuda", dtype=torch.float32)
+ router = CanonicalPRouter(RouterConfig()).to("cuda")
+ byte_encoder = ByteEntityEncoder(128)
+ wrapper = PythiaSplitTranslatedModel(base, package, router, byte_encoder).to("cuda").train()
+ representation.eval()
+ all_train_surfaces, all_heldout_surfaces = entity_split()
+ split_rng = random.Random(seed + 1)
+ train_surfaces = split_rng.sample(all_train_surfaces, 256)
+ required = ["silver key", "gold key"]
+ train_adjectives = {surface.split()[0] for surface in train_surfaces}
+ train_nouns = {surface.split()[1] for surface in train_surfaces}
+ compositional_heldout = [
+ surface for surface in all_heldout_surfaces
+ if surface.split()[0] in train_adjectives
+ and surface.split()[1] in train_nouns
+ and surface not in required
+ ]
+ heldout_surfaces = required + compositional_heldout[:62]
+ train_surface_set = set(train_surfaces)
+ assert not train_surface_set.intersection(heldout_surfaces)
+
+ encoded_values = [tokenizer.encode(" " + value, add_special_tokens=False) for value in VALUE_LABELS]
+ if any(len(ids) != 1 for ids in encoded_values):
+ raise RuntimeError("controlled values must be single Pythia tokens")
+ value_token_ids = [ids[0] for ids in encoded_values]
+ lm_values = base.get_output_embeddings().weight[value_token_ids].detach().float()
+ normalized_lm_values = F.normalize(lm_values, dim=-1)
+
+ def query_texts(surfaces, relations, leads):
+ texts, names, relation_ids = [], [], []
+ for lead in leads:
+ for surface, relation in zip(surfaces, relations):
+ texts.append(lead + RELATION_PROMPTS[relation].format(entity=surface))
+ names.append(surface)
+ relation_ids.append(relation)
+ return texts, names, relation_ids
+
+ def tokenize(texts):
+ return tokenizer(
+ texts, return_tensors="pt", padding=True, add_special_tokens=False
+ ).to("cuda")
+
+ def capture(texts, chunk=24):
+ by_layer = {layer: [] for layer in layers}
+ handles = [
+ base.gpt_neox.layers[layer].register_forward_hook(
+ lambda _module, _inputs, output, layer=layer: by_layer[layer].append(
+ output[:, -1].detach().cpu()
+ )
+ )
+ for layer in layers
+ ]
+ for start in range(0, len(texts), chunk):
+ with torch.inference_mode():
+ wrapper(**tokenize(texts[start:start + chunk]), use_cache=False)
+ for handle in handles:
+ handle.remove()
+ return torch.stack([
+ torch.cat(by_layer[layer]).to("cuda") for layer in layers
+ ])
+
+ train_relations = [index % 3 for index in range(len(train_surfaces))]
+ train_texts, train_names, train_relation_ids = query_texts(
+ train_surfaces, train_relations, TRAIN_LEADS
+ )
+ train_hidden = capture(train_texts)
+ train_names = train_names
+ train_relation_ids = torch.tensor(train_relation_ids, device="cuda")
+ train_entity_targets = byte_encoder(train_names).to("cuda")
+ heldout_relations = [index % 3 for index in range(len(heldout_surfaces))]
+ heldout_texts, heldout_names, heldout_relation_ids = query_texts(
+ heldout_surfaces, heldout_relations, (HELDOUT_LEADS[0],)
+ )
+ heldout_hidden = capture(heldout_texts)
+ heldout_relation_ids = torch.tensor(heldout_relation_ids, device="cuda")
+ heldout_entity_targets = byte_encoder(heldout_names).to("cuda")
+
+ query_optimizer = torch.optim.AdamW(package.query_projector.parameters(), lr=2e-3, eps=1e-6)
+ query_losses = []
+ for _ in range(query_steps):
+ surface_indices = rng.sample(range(len(train_surfaces)), 32)
+ hidden = torch.stack([
+ train_hidden[rng.randrange(attachment_count), index]
+ for index in surface_indices
+ ])
+ projected = package.query_projector(hidden)
+ targets = byte_encoder([train_surfaces[index] for index in surface_indices]).to("cuda")
+ relations = torch.tensor(
+ [train_relations[index] for index in surface_indices], device="cuda"
+ )
+ entity_loss = 1 - F.cosine_similarity(projected.entity, targets, dim=-1).mean()
+ contrastive = F.cross_entropy(
+ projected.entity @ targets.T / 0.07,
+ torch.arange(len(surface_indices), device="cuda"),
+ )
+ relation_loss = F.cross_entropy(projected.relation_logits, relations)
+ metadata_loss = F.cross_entropy(
+ projected.metadata_logits,
+ torch.zeros(len(surface_indices), dtype=torch.long, device="cuda"),
+ )
+ loss = entity_loss + contrastive + relation_loss + 0.25 * metadata_loss
+ query_optimizer.zero_grad(set_to_none=True)
+ loss.backward()
+ query_optimizer.step()
+ query_losses.append(float(loss.detach()))
+ del query_optimizer
+
+ def query_metrics(hidden, names, relations):
+ with torch.inference_mode():
+ projected = package.query_projector(hidden.mean(0))
+ targets = byte_encoder(names).to("cuda")
+ entity_scores = projected.entity @ targets.T
+ return {
+ "entity_accuracy": float((
+ entity_scores.argmax(-1) == torch.arange(len(names), device="cuda")
+ ).float().mean()),
+ "relation_accuracy": float((
+ projected.relation_logits.argmax(-1) == relations
+ ).float().mean()),
+ "metadata_accuracy": float((
+ projected.metadata_logits.argmax(-1) == 0
+ ).float().mean()),
+ "entity_cosine": float(F.cosine_similarity(
+ projected.entity, targets, dim=-1
+ ).mean()),
+ }
+
+ query_heldout_metrics = query_metrics(
+ heldout_hidden, heldout_names, heldout_relation_ids
+ )
+ byte_surface_metrics = {
+ "entity_accuracy": 1.0,
+ "relation_accuracy": query_heldout_metrics["relation_accuracy"],
+ "metadata_accuracy": query_heldout_metrics["metadata_accuracy"],
+ "entity_cosine": 1.0,
+ "tokenizer_independent": True,
+ "oracle_slot_assignments": 0,
+ }
+
+ lexical_projector = FrozenLexicalAnchorProjector(int(base.config.hidden_size)).to("cuda")
+ embedding = base.get_input_embeddings().weight.detach()
+
+ def lexical(surfaces):
+ values = []
+ for surface in surfaces:
+ ids = tokenizer.encode(" " + surface, add_special_tokens=False)
+ values.append(embedding[torch.tensor(ids, device="cuda")].float().mean(0))
+ return torch.stack(values)
+
+ train_lexical = lexical(train_surfaces)
+ lexical_optimizer = torch.optim.AdamW(lexical_projector.parameters(), lr=2e-3)
+ for _ in range(query_steps):
+ indices = torch.tensor(rng.sample(range(len(train_surfaces)), 32), device="cuda")
+ output = lexical_projector(train_lexical.index_select(0, indices))
+ target = byte_encoder([train_surfaces[int(index)] for index in indices]).to("cuda")
+ loss = 1 - F.cosine_similarity(output, target, dim=-1).mean()
+ lexical_optimizer.zero_grad(set_to_none=True)
+ loss.backward()
+ lexical_optimizer.step()
+ with torch.inference_mode():
+ lexical_output = lexical_projector(lexical(heldout_surfaces))
+ lexical_targets = byte_encoder(heldout_surfaces).to("cuda")
+ lexical_metrics = {
+ "entity_accuracy": float((
+ (lexical_output @ lexical_targets.T).argmax(-1)
+ == torch.arange(len(heldout_surfaces), device="cuda")
+ ).float().mean()),
+ "entity_cosine": float(F.cosine_similarity(
+ lexical_output, lexical_targets, dim=-1
+ ).mean()),
+ }
+ del lexical_optimizer, lexical_projector, train_lexical
+
+ for parameter in package.query_projector.parameters():
+ parameter.requires_grad_(False)
+ router_optimizer = torch.optim.AdamW(router.parameters(), lr=1e-2)
+ router_losses = []
+ for _ in range(router_steps):
+ source_index = rng.randrange(len(train_names))
+ correct_surface = train_names[source_index]
+ relation = int(train_relation_ids[source_index])
+ relation_logits = torch.full((1, 3), -12.0, device="cuda")
+ relation_logits[0, relation] = 12.0
+ query = FactorizedCanonicalQuery(
+ entity=byte_encoder([correct_surface]).to("cuda"),
+ relation_logits=relation_logits,
+ metadata_logits=torch.tensor([[12.0, -12.0, -12.0, -12.0]], device="cuda"),
+ )
+ candidates = [correct_surface, rng.choice(train_surfaces)]
+ while candidates[1] == correct_surface:
+ candidates[1] = rng.choice(train_surfaces)
+ candidates.extend((correct_surface, correct_surface))
+ candidate_relations = [relation, relation, (relation + 1) % 3, relation]
+ candidate_metadata = [CANONICAL, CANONICAL, CANONICAL, HISTORICAL]
+ while len(candidates) < 128:
+ candidates.append(rng.choice(train_surfaces))
+ candidate_relations.append(rng.randrange(3))
+ candidate_metadata.append(CANONICAL)
+ permutation = list(range(len(candidates)))
+ rng.shuffle(permutation)
+ candidates = [candidates[index] for index in permutation]
+ index = CanonicalRouterIndex(
+ entity=byte_encoder(candidates).to("cuda"),
+ relation_id=torch.tensor([candidate_relations[i] for i in permutation], device="cuda"),
+ metadata_id=torch.tensor([candidate_metadata[i] for i in permutation], device="cuda"),
+ valid=torch.ones(len(candidates), dtype=torch.bool, device="cuda"),
+ )
+ target = torch.tensor([permutation.index(0)], device="cuda")
+ scores, _ = router.all_scores(query, index)
+ labels = torch.zeros_like(scores)
+ labels[:, target] = 1.0
+ loss = F.cross_entropy(scores, target) + F.binary_cross_entropy_with_logits(
+ scores, labels, pos_weight=torch.tensor([len(candidates) - 1.0], device="cuda")
+ )
+ router_optimizer.zero_grad(set_to_none=True)
+ loss.backward()
+ router_optimizer.step()
+ router_losses.append(float(loss.detach()))
+ del router_optimizer
+ calibration_positive = []
+ calibration_negative = []
+ with torch.inference_mode():
+ for calibration_index in range(128):
+ surface = train_surfaces[calibration_index]
+ relation = calibration_index % 3
+ wrong = train_surfaces[(calibration_index + 37) % len(train_surfaces)]
+ query = FactorizedCanonicalQuery(
+ entity=byte_encoder([surface]).to("cuda"),
+ relation_logits=torch.full((1, 3), -12.0, device="cuda"),
+ metadata_logits=torch.tensor([[12.0, -12.0, -12.0, -12.0]], device="cuda"),
+ )
+ query.relation_logits[0, relation] = 12.0
+ index = CanonicalRouterIndex(
+ entity=byte_encoder([surface, wrong, surface, surface]).to("cuda"),
+ relation_id=torch.tensor([relation, relation, (relation + 1) % 3, relation], device="cuda"),
+ metadata_id=torch.tensor([CANONICAL, CANONICAL, CANONICAL, HISTORICAL], device="cuda"),
+ valid=torch.ones(4, dtype=torch.bool, device="cuda"),
+ )
+ scores, _ = router.all_scores(query, index)
+ calibration_positive.append(scores[0, 0])
+ calibration_negative.extend(scores[0, 1:])
+ calibration_balanced_accuracy = router.calibrate_acceptance(
+ torch.stack(calibration_positive), torch.stack(calibration_negative)
+ )
+ for parameter in router.parameters():
+ parameter.requires_grad_(False)
+ for parameter in package.query_projector.parameters():
+ parameter.requires_grad_(True)
+
+ def canonical_vector(entity_id, relation, value, metadata=CANONICAL):
+ with torch.inference_mode():
+ vector = representation.encode(
+ torch.tensor([entity_id % 24]), torch.tensor([relation]),
+ torch.tensor([value % 36]), torch.tensor([metadata]),
+ )[0]
+ return vector.to("cuda", dtype=torch.float16)
+
+ value_optimizer = torch.optim.AdamW(package.value_translator.parameters(), lr=2e-3, eps=1e-6)
+ value_losses = []
+ for _ in range(value_steps):
+ entity_ids = [rng.randrange(24) for _ in range(32)]
+ relations = [rng.randrange(3) for _ in range(32)]
+ values = [rng.randrange(36) for _ in range(32)]
+ canonical = torch.stack([
+ canonical_vector(entity, relation, value).float()
+ for entity, relation, value in zip(entity_ids, relations, values)
+ ])
+ translated = package.value_translator(canonical)
+ normalized = F.normalize(translated, dim=-1)
+ targets = torch.tensor(values, device="cuda")
+ loss = (
+ 1 - F.cosine_similarity(
+ normalized, normalized_lm_values.index_select(0, targets), dim=-1
+ ).mean()
+ + F.cross_entropy(normalized @ normalized_lm_values.T / 0.07, targets)
+ )
+ value_optimizer.zero_grad(set_to_none=True)
+ loss.backward()
+ value_optimizer.step()
+ value_losses.append(float(loss.detach()))
+ del value_optimizer
+ with torch.inference_mode():
+ heldout_value_ids = torch.arange(36, device="cuda")
+ heldout_vectors = torch.stack([
+ canonical_vector(index % 24, index % 3, index).float() for index in range(36)
+ ])
+ heldout_value_output = F.normalize(
+ package.value_translator(heldout_vectors), dim=-1
+ )
+ value_metrics = {
+ "accuracy": float((
+ (heldout_value_output @ normalized_lm_values.T).argmax(-1)
+ == heldout_value_ids
+ ).float().mean()),
+ "cosine": float(F.cosine_similarity(
+ heldout_value_output, normalized_lm_values, dim=-1
+ ).mean()),
+ }
+
+ rp_preserve = tokenize([
+ "A patient tailor compared blue ribbons while rain ticked softly against the shop window.",
+ "Two actors rehearsed a harmless joke and rearranged wooden chairs beside the empty stage.",
+ ])
+ rp_eval = tokenize([
+ "At dusk, a baker swept flour from the counter while neighbors debated tomorrow's parade.",
+ "A sleepy musician closed the balcony doors and described clouds drifting above the orchard.",
+ ])
+ with torch.inference_mode():
+ frozen_rp_preserve = wrapper(**rp_preserve, use_cache=False).logits.detach()
+ frozen_rp_eval = wrapper(**rp_eval, use_cache=False).logits.detach()
+ base_rp_loss = float(wrapper(**rp_eval, labels=rp_eval.input_ids, use_cache=False).loss)
+
+ train_state_surfaces = train_surfaces[:24]
+ heldout_state_surfaces = heldout_surfaces[:20]
+ train_value_assignment = {surface: index % 36 for index, surface in enumerate(train_state_surfaces)}
+ heldout_value_assignment = {surface: (index * 5 + 3) % 36 for index, surface in enumerate(heldout_state_surfaces)}
+ train_owner_hidden = capture([
+ TRAIN_LEADS[0] + RELATION_PROMPTS[0].format(entity=surface)
+ for surface in train_state_surfaces
+ ]).mean(0)
+ heldout_owner_hidden = capture([
+ HELDOUT_LEADS[1] + RELATION_PROMPTS[0].format(entity=surface)
+ for surface in heldout_state_surfaces
+ ]).mean(0)
+ with torch.inference_mode():
+ owner_queries = package.query_projector(
+ heldout_owner_hidden,
+ entity_anchor=byte_encoder(heldout_state_surfaces).to("cuda"),
+ )
+ hard_top1 = []
+ false_positives = {
+ "wrong_entity": [], "wrong_relation": [], "historical": [],
+ "invalidated": [], "irrelevant": [],
+ }
+ for query_index, surface in enumerate(heldout_state_surfaces):
+ wrong_entity = heldout_state_surfaces[(query_index + 1) % len(heldout_state_surfaces)]
+ candidates = [surface, wrong_entity, surface, surface, f"irrelevant {query_index}"]
+ index = CanonicalRouterIndex(
+ entity=byte_encoder(candidates).to("cuda"),
+ relation_id=torch.tensor([0, 0, 1, 0, 2], device="cuda"),
+ metadata_id=torch.tensor([CANONICAL, CANONICAL, CANONICAL, HISTORICAL, CANONICAL], device="cuda"),
+ valid=torch.ones(5, dtype=torch.bool, device="cuda"),
+ )
+ single_query = FactorizedCanonicalQuery(
+ entity=owner_queries.entity[query_index:query_index + 1],
+ relation_logits=owner_queries.relation_logits[query_index:query_index + 1],
+ metadata_logits=owner_queries.metadata_logits[query_index:query_index + 1],
+ )
+ scores, _ = router.all_scores(single_query, index)
+ hard_top1.append(int(scores.argmax(-1)) == 0)
+ for label, candidate_index in (
+ ("wrong_entity", 1), ("wrong_relation", 2),
+ ("historical", 3), ("irrelevant", 4),
+ ):
+ false_positives[label].append(
+ float(scores[0, candidate_index]) >= float(router.acceptance_threshold)
+ )
+ invalid_index = CanonicalRouterIndex(
+ entity=index.entity[:1], relation_id=index.relation_id[:1],
+ metadata_id=index.metadata_id[:1],
+ valid=torch.zeros(1, dtype=torch.bool, device="cuda"),
+ )
+ invalid_scores, _ = router.all_scores(single_query, invalid_index)
+ false_positives["invalidated"].append(bool(torch.isfinite(invalid_scores[0, 0])))
+ router_hard_metrics = {
+ "top1_accuracy": sum(hard_top1) / len(hard_top1),
+ **{
+ f"{label}_false_positive_rate": sum(values) / len(values)
+ for label, values in false_positives.items()
+ },
+ }
+
+ def make_store(entries, slots=None, local_rng=None, metadata=CANONICAL):
+ capacity = slots or max(4, len(entries))
+ store = CanonicalPStore(CanonicalPConfig(
+ slots=capacity, width=512, dtype=torch.float16, device="cuda", merge_similarity=1.0
+ ))
+ rows = list(entries)
+ if local_rng:
+ local_rng.shuffle(rows)
+ slot_by_surface = {}
+ for ordinal, (surface, relation, value) in enumerate(rows):
+ slot, _ = store.create(
+ canonical_vector(ordinal, relation, value, metadata),
+ entity_id=ordinal, relation_id=relation, value_id=value, metadata_id=metadata,
+ label=surface,
+ )
+ slot_by_surface[surface] = slot
+ return store, slot_by_surface
+
+ def state_inputs(surfaces, lead=TRAIN_LEADS[0], relation=0):
+ return tokenize([
+ lead + RELATION_PROMPTS[relation].format(entity=surface) for surface in surfaces
+ ])
+
+ training_base_logits = {}
+ for lead in TRAIN_LEADS:
+ with torch.inference_mode():
+ training_base_logits[lead] = wrapper(
+ **state_inputs(train_state_surfaces, lead), use_cache=False
+ ).logits[:, -1].detach()
+
+ package_optimizer = torch.optim.AdamW(package.parameters(), lr=5e-4, eps=1e-6)
+ causal_losses = []
+ pre_preservation_ablation = None
+ midpoint_entries = [
+ (surface, 0, heldout_value_assignment[surface])
+ for surface in heldout_state_surfaces
+ ]
+ midpoint_store, _ = make_store(
+ midpoint_entries, slots=128, local_rng=random.Random(seed + 800)
+ )
+ midpoint_inputs = state_inputs(heldout_state_surfaces, HELDOUT_LEADS[1])
+ midpoint_expected = torch.tensor(
+ [heldout_value_assignment[surface] for surface in heldout_state_surfaces],
+ device="cuda",
+ )
+ for step in range(causal_steps):
+ surfaces = rng.sample(train_state_surfaces, 8)
+ entries = [(surface, 0, train_value_assignment[surface]) for surface in surfaces]
+ store, _ = make_store(entries, slots=128, local_rng=rng)
+ lead = rng.choice(TRAIN_LEADS)
+ inputs = state_inputs(surfaces, lead)
+ targets = torch.tensor(
+ [value_token_ids[train_value_assignment[surface]] for surface in surfaces],
+ device="cuda",
+ )
+ output = wrapper(
+ **inputs, p_store=store, query_entity_surfaces=surfaces, use_cache=False
+ )
+ state_loss = F.cross_entropy(output.logits[:, -1].float(), targets)
+
+ wrong_entries = [
+ (rng.choice(train_surfaces[len(train_state_surfaces):]), 0,
+ train_value_assignment[surface])
+ for surface in surfaces
+ ]
+ wrong_store, _ = make_store(wrong_entries, slots=128)
+ wrong = wrapper(
+ **inputs, p_store=wrong_store, query_entity_surfaces=surfaces, use_cache=False
+ ).logits[:, -1].float()
+ historical_store, _ = make_store(entries, slots=128, metadata=HISTORICAL)
+ historical = wrapper(
+ **inputs, p_store=historical_store,
+ query_entity_surfaces=surfaces, use_cache=False,
+ ).logits[:, -1].float()
+ indices = torch.tensor([train_state_surfaces.index(surface) for surface in surfaces], device="cuda")
+ base_logits = training_base_logits[lead].index_select(0, indices).float()
+ wrong_preserve = F.kl_div(
+ F.log_softmax(base_logits, dim=-1), F.softmax(wrong, dim=-1), reduction="batchmean"
+ )
+ historical_preserve = F.kl_div(
+ F.log_softmax(base_logits, dim=-1),
+ F.softmax(historical, dim=-1), reduction="batchmean"
+ )
+
+ hidden_indices = torch.tensor([
+ train_state_surfaces.index(surface) for surface in surfaces
+ ], device="cuda")
+ projected = package.query_projector(train_owner_hidden.index_select(0, hidden_indices))
+ query_loss = (
+ 1 - F.cosine_similarity(
+ projected.entity, byte_encoder(surfaces).to("cuda"), dim=-1
+ ).mean()
+ + F.cross_entropy(projected.relation_logits, torch.zeros(8, dtype=torch.long, device="cuda"))
+ )
+ vectors = torch.stack([
+ canonical_vector(index, 0, train_value_assignment[surface]).float()
+ for index, surface in enumerate(surfaces)
+ ])
+ translated = F.normalize(package.value_translator(vectors), dim=-1)
+ value_targets = torch.tensor(
+ [train_value_assignment[surface] for surface in surfaces], device="cuda"
+ )
+ value_loss = 1 - F.cosine_similarity(
+ translated, normalized_lm_values.index_select(0, value_targets), dim=-1
+ ).mean()
+
+ loss = (
+ state_loss + 0.5 * query_loss + 0.2 * value_loss
+ + 2.0 * wrong_preserve + 2.0 * historical_preserve
+ )
+ if step >= causal_steps // 2:
+ rp_output = wrapper(
+ **rp_preserve, labels=rp_preserve.input_ids, p_store=store, use_cache=False
+ )
+ rp_kl = F.kl_div(
+ F.log_softmax(frozen_rp_preserve[:, -1].float(), dim=-1),
+ F.softmax(rp_output.logits[:, -1].float(), dim=-1), reduction="batchmean"
+ )
+ loss = loss + 0.05 * rp_output.loss.float() + 2.0 * rp_kl
+ package_optimizer.zero_grad(set_to_none=True)
+ loss.backward()
+ torch.nn.utils.clip_grad_norm_(package.parameters(), 1.0)
+ package_optimizer.step()
+ causal_losses.append(float(state_loss.detach()))
+ if step + 1 == causal_steps // 2:
+ with torch.inference_mode():
+ midpoint_logits = wrapper(
+ **midpoint_inputs, p_store=midpoint_store,
+ query_entity_surfaces=heldout_state_surfaces, use_cache=False,
+ ).logits[:, -1].float()
+ midpoint_rp = wrapper(
+ **rp_eval, labels=rp_eval.input_ids,
+ p_store=midpoint_store, use_cache=False,
+ )
+ midpoint_rp_kl = F.kl_div(
+ F.log_softmax(frozen_rp_eval[:, -1].float(), dim=-1),
+ F.softmax(midpoint_rp.logits[:, -1].float(), dim=-1),
+ reduction="batchmean",
+ )
+ pre_preservation_ablation = {
+ "state_loss": float(state_loss.detach()),
+ "state_candidate_accuracy": float((
+ midpoint_logits[:, value_token_ids].argmax(-1) == midpoint_expected
+ ).float().mean()),
+ "wrong_state_kl": float(wrong_preserve.detach()),
+ "historical_state_kl": float(historical_preserve.detach()),
+ "rp_loss": float(midpoint_rp.loss),
+ "rp_kl": float(midpoint_rp_kl),
+ }
+ del package_optimizer
+ wrapper.eval()
+
+ def store_bytes(store):
+ tensors = (
+ store.cache.values, store.cache.valid, store.cache.slot_type,
+ store.cache.confidence, store.cache.importance, store.cache.freshness,
+ store.cache.persistence, store.cache.last_updated, store.cache.source,
+ store.entity_id, store.relation_id, store.value_id,
+ store.canonical_metadata_id,
+ )
+ return sum(t.numel() * t.element_size() for t in tensors)
+
+ def index_bytes(index):
+ return sum(
+ tensor.numel() * tensor.element_size()
+ for tensor in (index.entity, index.relation_id, index.metadata_id, index.valid)
+ )
+
+ scaling = {}
+ all_distractors = [
+ surface for surface in train_surfaces + heldout_surfaces
+ if surface not in heldout_state_surfaces
+ ]
+ while len(all_distractors) < 512:
+ all_distractors.append(f"irrelevant entity {len(all_distractors)}")
+ for slot_count in SLOT_SIZES:
+ query_count = min(20, slot_count)
+ query_surfaces = heldout_state_surfaces[:query_count]
+ entries = [
+ (surface, 0, heldout_value_assignment[surface]) for surface in query_surfaces
+ ]
+ for index in range(slot_count - query_count):
+ entries.append((all_distractors[index], (index + 1) % 3, (index + 7) % 36))
+ store, slot_map = make_store(entries, slots=slot_count, local_rng=random.Random(seed + slot_count))
+ inputs = state_inputs(query_surfaces, HELDOUT_LEADS[1])
+ hidden = capture([
+ HELDOUT_LEADS[1] + RELATION_PROMPTS[0].format(entity=surface)
+ for surface in query_surfaces
+ ]).mean(0)
+ with torch.inference_mode():
+ hidden_only_query = package.query_projector(hidden)
+ query = package.query_projector(
+ hidden, entity_anchor=byte_encoder(query_surfaces).to("cuda")
+ )
+ index = router.build_index(store, byte_encoder, device="cuda")
+ scores, _ = router.all_scores(query, index)
+ expected = torch.tensor([slot_map[surface] for surface in query_surfaces], device="cuda")
+ order = scores.argsort(dim=-1, descending=True)
+ ranks = (order == expected[:, None]).nonzero()[:, 1] + 1
+ hidden_only_scores, _ = router.all_scores(hidden_only_query, index)
+ hidden_only_order = hidden_only_scores.argsort(dim=-1, descending=True)
+ hidden_only_ranks = (
+ hidden_only_order == expected[:, None]
+ ).nonzero()[:, 1] + 1
+ oracle_query = FactorizedCanonicalQuery(
+ entity=byte_encoder(query_surfaces).to("cuda"),
+ relation_logits=torch.tensor([[12.0, -12.0, -12.0]], device="cuda").expand(query_count, -1),
+ metadata_logits=torch.tensor([[12.0, -12.0, -12.0, -12.0]], device="cuda").expand(query_count, -1),
+ )
+ oracle_scores, _ = router.all_scores(oracle_query, index)
+ oracle_order = oracle_scores.argsort(dim=-1, descending=True)
+ oracle_ranks = (oracle_order == expected[:, None]).nonzero()[:, 1] + 1
+ route_metrics = {
+ "top1_accuracy": float((ranks == 1).float().mean()),
+ "top2_recall": float((ranks <= 2).float().mean()),
+ "top4_recall": float((ranks <= 4).float().mean()),
+ "mrr": float((1.0 / ranks.float()).mean()),
+ "hidden_only_top1_accuracy": float((hidden_only_ranks == 1).float().mean()),
+ "oracle_query_top1_accuracy": float((oracle_ranks == 1).float().mean()),
+ "oracle_query_top4_recall": float((oracle_ranks <= 4).float().mean()),
+ "oracle_query_mrr": float((1.0 / oracle_ranks.float()).mean()),
+ }
+ logits = wrapper(
+ **inputs, p_store=store, query_entity_surfaces=query_surfaces,
+ use_cache=False,
+ ).logits[:, -1].float()
+ expected_values = torch.tensor(
+ [heldout_value_assignment[surface] for surface in query_surfaces], device="cuda"
+ )
+ generation_accuracy = float((
+ logits[:, value_token_ids].argmax(-1) == expected_values
+ ).float().mean())
+ for _ in range(2):
+ wrapper(
+ **inputs, p_store=store, query_entity_surfaces=query_surfaces,
+ use_cache=False,
+ )
+ torch.cuda.synchronize()
+ start = time.perf_counter()
+ for _ in range(5):
+ wrapper(
+ **inputs, p_store=store, query_entity_surfaces=query_surfaces,
+ use_cache=False,
+ )
+ torch.cuda.synchronize()
+ latency = (time.perf_counter() - start) / 5
+ scaling[str(slot_count)] = {
+ **route_metrics,
+ "state_generation_accuracy": generation_accuracy,
+ "latency_seconds": latency,
+ "active_vram_overhead_bytes": (
+ _bytes(package.parameters()) + _bytes(router.parameters())
+ + store_bytes(store) + index_bytes(index)
+ ),
+ }
+
+ eval_entries = [
+ (surface, 0, heldout_value_assignment[surface]) for surface in heldout_state_surfaces
+ ]
+ eval_store, eval_slots = make_store(eval_entries, slots=128, local_rng=random.Random(seed + 900))
+ eval_inputs = state_inputs(heldout_state_surfaces, HELDOUT_LEADS[1])
+ expected_values = torch.tensor(
+ [heldout_value_assignment[surface] for surface in heldout_state_surfaces], device="cuda"
+ )
+ expected_tokens = torch.tensor([value_token_ids[int(value)] for value in expected_values], device="cuda")
+ oracle_indices = torch.tensor([eval_slots[surface] for surface in heldout_state_surfaces], device="cuda")
+ with torch.inference_mode():
+ disabled = wrapper(**eval_inputs, use_cache=False).logits[:, -1].float()
+ oracle = wrapper(
+ **eval_inputs, p_store=eval_store, oracle_indices=oracle_indices,
+ query_entity_surfaces=heldout_state_surfaces,
+ gate_enabled=False, use_cache=False,
+ ).logits[:, -1].float()
+ without_gate = wrapper(
+ **eval_inputs, p_store=eval_store,
+ query_entity_surfaces=heldout_state_surfaces,
+ gate_enabled=False, use_cache=False
+ ).logits[:, -1].float()
+ full = wrapper(
+ **eval_inputs, p_store=eval_store,
+ query_entity_surfaces=heldout_state_surfaces,
+ collect_telemetry=True, use_cache=False
+ ).logits[:, -1].float()
+ full_gate = float(torch.stack([
+ values[:, -1].float().mean() for values in wrapper.gate_telemetry
+ ]).mean())
+
+ def accuracy(logits):
+ return float((logits[:, value_token_ids].argmax(-1) == expected_values).float().mean())
+
+ ablations = {
+ "router_only": scaling["128"],
+ "translator_only_oracle_routing": {"state_candidate_accuracy": accuracy(oracle)},
+ "router_plus_translator_without_gate": {"state_candidate_accuracy": accuracy(without_gate)},
+ "router_plus_translator_plus_gate": pre_preservation_ablation,
+ "full_system_with_preservation": {
+ "state_candidate_accuracy": accuracy(full),
+ "full_token_accuracy": float((full.argmax(-1) == expected_tokens).float().mean()),
+ "gate_activation": full_gate,
+ },
+ }
+
+ def single_condition(surface, value, *, label=None, metadata=CANONICAL, invalidate=False):
+ store, slots = make_store(
+ [(label or surface, 0, value)], slots=128, metadata=metadata
+ )
+ if invalidate:
+ store.invalidate(next(iter(slots.values())))
+ return store
+
+ counter_prompt = state_inputs(["silver key"], HELDOUT_LEADS[1])
+ alice, bob = 0, 1
+ conditions = {
+ "disabled": None,
+ "p1_silver_alice": single_condition("silver key", alice),
+ "p2_silver_bob": single_condition("silver key", bob),
+ "p3_gold_alice": single_condition("silver key", alice, label="gold key"),
+ "p4_silver_historical": single_condition("silver key", alice, metadata=HISTORICAL),
+ "p4_silver_invalidated": single_condition("silver key", alice, invalidate=True),
+ }
+ counterfactual = {}
+ with torch.inference_mode():
+ for label, store in conditions.items():
+ output = wrapper(
+ **counter_prompt, p_store=store,
+ query_entity_surfaces=["silver key"],
+ collect_telemetry=True, use_cache=False
+ ).logits[:, -1].float()
+ probabilities = F.softmax(output, dim=-1)
+ counterfactual[label] = {
+ "alice_logit": float(output[0, value_token_ids[alice]]),
+ "bob_logit": float(output[0, value_token_ids[bob]]),
+ "alice_probability": float(probabilities[0, value_token_ids[alice]]),
+ "bob_probability": float(probabilities[0, value_token_ids[bob]]),
+ "generated": tokenizer.decode([int(output.argmax(-1))]),
+ "gate": 0.0 if not wrapper.gate_telemetry else float(torch.stack([
+ values[:, -1].float().mean() for values in wrapper.gate_telemetry
+ ]).mean()),
+ }
+
+ wrong_store = conditions["p3_gold_alice"]
+ invalid_store = conditions["p4_silver_invalidated"]
+
+ def greedy_continuations(inputs, store, steps=6):
+ input_ids = inputs.input_ids.clone()
+ attention_mask = inputs.attention_mask.clone()
+ original_length = input_ids.shape[1]
+ for _ in range(steps):
+ output = wrapper(
+ input_ids=input_ids, attention_mask=attention_mask,
+ p_store=store, use_cache=False,
+ ).logits[:, -1]
+ next_token = output.argmax(-1, keepdim=True)
+ input_ids = torch.cat((input_ids, next_token), dim=1)
+ attention_mask = torch.cat((
+ attention_mask,
+ torch.ones_like(next_token, dtype=attention_mask.dtype),
+ ), dim=1)
+ return [tokenizer.decode(row[original_length:]) for row in input_ids]
+
+ with torch.inference_mode():
+ rp_conditions = {}
+ for label, store in (
+ ("base", None), ("irrelevant", eval_store), ("wrong_entity", wrong_store),
+ ("invalidated", invalid_store),
+ ):
+ output = wrapper(
+ **rp_eval, labels=rp_eval.input_ids, p_store=store, use_cache=False
+ )
+ kl = F.kl_div(
+ F.log_softmax(frozen_rp_eval[:, -1].float(), dim=-1),
+ F.softmax(output.logits[:, -1].float(), dim=-1), reduction="batchmean"
+ )
+ rp_conditions[label] = {
+ "loss": float(output.loss), "kl": float(kl),
+ "samples": greedy_continuations(rp_eval, store),
+ }
+
+ invalid_difference = max(
+ abs(counterfactual["p4_silver_invalidated"][key] - counterfactual["disabled"][key])
+ for key in ("alice_logit", "bob_logit")
+ )
+ mutation_store, mutation_slots = make_store(
+ [("silver key", 0, 0)], slots=128
+ )
+ mutation_slot = mutation_slots["silver key"]
+ for mutation_index, value in enumerate((1, 2, 3, 4)):
+ mutation_store.modify(
+ mutation_slot,
+ canonical_vector(0, 0, value),
+ entity_id=0, relation_id=0, value_id=value, metadata_id=CANONICAL,
+ source=SlotSource.CORRECTION if mutation_index == 3 else None,
+ )
+ with torch.inference_mode():
+ mutation_logits = wrapper(
+ **counter_prompt, p_store=mutation_store,
+ query_entity_surfaces=["silver key"], use_cache=False,
+ ).logits[:, -1].float()
+ mutation_latest_correct = int(
+ mutation_logits[:, value_token_ids].argmax(-1)
+ ) == 4
+ mutation_store.invalidate(mutation_slot)
+ with torch.inference_mode():
+ mutation_invalidated = wrapper(
+ **counter_prompt, p_store=mutation_store,
+ query_entity_surfaces=["silver key"], use_cache=False,
+ ).logits[:, -1].float()
+ mutation_disabled = wrapper(**counter_prompt, use_cache=False).logits[:, -1].float()
+ mutation_invalidated_difference = float((
+ mutation_invalidated - mutation_disabled
+ ).abs().max())
+ base_gradients = sum(parameter.grad is not None for parameter in base.parameters())
+ if package_path is not None:
+ package.save(package_path)
+ restored = SplitPTranslatePackage.load(package_path, device="cuda")
+ restored.validate_compatibility(
+ model_id=pythia_model_identifier(base), model_hidden_width=int(base.config.hidden_size),
+ attachment_layers=layers, model_config_sha256=config_checksum(base.config),
+ )
+ package_roundtrip = max(
+ float((left - right).abs().max())
+ for left, right in zip(package.state_dict().values(), restored.state_dict().values())
+ )
+ del restored
+ else:
+ package_roundtrip = None
+ if router_path is not None:
+ router.save(router_path)
+ restored_router = CanonicalPRouter.load(router_path, device="cuda")
+ router_roundtrip = max(
+ float((left - right).abs().max())
+ for left, right in zip(router.state_dict().values(), restored_router.state_dict().values())
+ )
+ del restored_router
+ else:
+ router_roundtrip = None
+
+ result = {
+ "attachment_layers": list(layers),
+ "attachment_count": attachment_count,
+ "query_projector": {
+ "loss_first_last": [query_losses[0], query_losses[-1]],
+ "byte_surface_anchor_approach": byte_surface_metrics,
+ "hidden_to_byte_reconstruction_ablation": query_heldout_metrics,
+ "frozen_lexical_anchor_approach": lexical_metrics,
+ "heldout_names": len(heldout_surfaces),
+ "training_names": len(train_surfaces),
+ },
+ "router": {
+ "loss_first_last": [router_losses[0], router_losses[-1]],
+ "model_hidden_dimensions": 0,
+ "acceptance_threshold": float(router.acceptance_threshold),
+ "calibration_balanced_accuracy": calibration_balanced_accuracy,
+ "hard_negative_metrics": router_hard_metrics,
+ "scaling": scaling,
+ },
+ "value_translator": {
+ "loss_first_last": [value_losses[0], value_losses[-1]],
+ "oracle_selected_metrics": value_metrics,
+ },
+ "causal_training_loss_first_last": [causal_losses[0], causal_losses[-1]],
+ "ablations": ablations,
+ "counterfactual": counterfactual,
+ "invalidated_logit_difference": invalid_difference,
+ "mutation_chain": {
+ "latest_state_accuracy": float(mutation_latest_correct),
+ "invalidated_max_logit_difference": mutation_invalidated_difference,
+ "source_tokens_in_recent_kv": 0,
+ },
+ "natural_rp": {
+ "conditions": rp_conditions,
+ "base_loss": base_rp_loss,
+ "relevant_state_generation_sample": counterfactual["p1_silver_alice"]["generated"],
+ },
+ "base_parameters_with_grad": base_gradients,
+ "source_tokens_in_recent_kv": 0,
+ "extra_prompt_tokens": 0,
+ "package_parameters": sum(p.numel() for p in package.parameters()),
+ "router_parameters": sum(p.numel() for p in router.parameters()),
+ "package_roundtrip_max_difference": package_roundtrip,
+ "router_roundtrip_max_difference": router_roundtrip,
+ "package_path": str(package_path) if package_path else None,
+ "router_path": str(router_path) if router_path else None,
+ }
+ wrapper.close()
+ del wrapper, base, package, router
+ gc.collect()
+ torch.cuda.empty_cache()
+ return result
diff --git a/src/pcm/planner/web_chat.py b/src/pcm/planner/web_chat.py
new file mode 100644
index 0000000000000000000000000000000000000000..58d1baf1c4479d589b1814804d59e1153a00f625
--- /dev/null
+++ b/src/pcm/planner/web_chat.py
@@ -0,0 +1,376 @@
+"""llama.cpp Web UI gateway for live Planner Cache conversations.
+
+The browser is presentation only. Every chat request terminates here and passes
+through :class:`PlannerChatSession` before the selected frozen-model runtime is
+called. This prevents the llama.cpp UI from bypassing canonical memory.
+"""
+
+from __future__ import annotations
+
+import argparse
+import copy
+from http import HTTPStatus
+from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer
+import json
+from pathlib import Path
+import sys
+import threading
+import time
+from urllib.parse import parse_qs, urlsplit
+import webbrowser
+
+from pcm.planner.chat_cli import PlannerChatSession, display_json, parser as chat_parser
+
+
+WEB_HELP = """Chat in the browser using the llama.cpp Web UI.
+
+Terminal commands:
+ /help show this help
+ /state show active canonical P-cache entries
+ /personality show promoted personality entries and the last retrieval
+ /events show recent Planner Cache events
+ /save checkpoint P-cache and P-package state
+ /quit save and stop the Web UI
+"""
+
+
+def parser() -> argparse.ArgumentParser:
+ result = chat_parser()
+ result.description = "Planner Cache through the llama.cpp Web UI"
+ result.add_argument("--web-host", default="127.0.0.1")
+ result.add_argument("--web-port", type=int, default=0)
+ result.add_argument("--web-ui-path", type=Path, required=True)
+ result.add_argument("--no-browser", action="store_true")
+ return result
+
+
+def _text_content(content: object) -> str:
+ if isinstance(content, str):
+ return content
+ if isinstance(content, list):
+ parts = []
+ for item in content:
+ if isinstance(item, dict) and item.get("type") in {"text", "input_text"}:
+ parts.append(str(item.get("text", "")))
+ return "\n".join(parts)
+ return str(content or "")
+
+
+def newest_user_message(payload: dict[str, object]) -> str:
+ messages = payload.get("messages")
+ if not isinstance(messages, list):
+ raise ValueError("messages must be an array")
+ for row in reversed(messages):
+ if isinstance(row, dict) and row.get("role") == "user":
+ message = _text_content(row.get("content")).strip()
+ if message:
+ return message
+ raise ValueError("a non-empty user message is required")
+
+
+def native_messages(payload: dict[str, object]) -> list[dict[str, object]]:
+ """Return an opaque structural copy for the model's native chat template."""
+ messages = payload.get("messages")
+ if not isinstance(messages, list):
+ raise ValueError("messages must be an array")
+ if not all(isinstance(row, dict) for row in messages):
+ raise ValueError("every message must be an object")
+ return copy.deepcopy(messages)
+
+
+def completion_payload(
+ *, model: str, text: str, input_tokens: int | None, output_tokens: int | None,
+) -> dict[str, object]:
+ return {
+ "id": "chatcmpl-planner-cache",
+ "object": "chat.completion",
+ "created": int(time.time()),
+ "model": model,
+ "choices": [{
+ "index": 0,
+ "message": {"role": "assistant", "content": text},
+ "finish_reason": "stop",
+ }],
+ "usage": {
+ "prompt_tokens": input_tokens or 0,
+ "completion_tokens": output_tokens or 0,
+ "total_tokens": (input_tokens or 0) + (output_tokens or 0),
+ },
+ }
+
+
+class PlannerWebGateway:
+ """Own the exact llama.cpp UI assets and the Planner Cache API boundary."""
+
+ def __init__(
+ self,
+ session: PlannerChatSession,
+ ui_path: Path,
+ *,
+ host: str = "127.0.0.1",
+ port: int = 0,
+ ) -> None:
+ if not (ui_path / "index.html").is_file():
+ raise FileNotFoundError(f"llama.cpp Web UI not found: {ui_path}")
+ self.session = session
+ self.ui_path = ui_path.resolve()
+ self.lock = threading.Lock()
+ self.httpd = ThreadingHTTPServer((host, port), self._handler())
+ self.thread: threading.Thread | None = None
+
+ @property
+ def address(self) -> tuple[str, int]:
+ host, port = self.httpd.server_address[:2]
+ return str(host), int(port)
+
+ @property
+ def url(self) -> str:
+ host, port = self.address
+ visible_host = "127.0.0.1" if host in {"0.0.0.0", "::"} else host
+ return f"http://{visible_host}:{port}/"
+
+ def _handler(self):
+ gateway = self
+
+ class Handler(SimpleHTTPRequestHandler):
+ server_version = "PlannerCacheWeb/1"
+
+ def __init__(self, *args: object, **kwargs: object) -> None:
+ super().__init__(*args, directory=str(gateway.ui_path), **kwargs)
+
+ def log_message(self, _format: str, *args: object) -> None:
+ return
+
+ def _json(self, value: object, status: int = 200) -> None:
+ body = json.dumps(value, ensure_ascii=False).encode("utf-8")
+ self.send_response(status)
+ self.send_header("Content-Type", "application/json; charset=utf-8")
+ self.send_header("Content-Length", str(len(body)))
+ self.send_header("Cache-Control", "no-store")
+ self.end_headers()
+ self.wfile.write(body)
+ self.wfile.flush()
+
+ def _read_json(self) -> dict[str, object]:
+ length = int(self.headers.get("Content-Length", "0"))
+ value = json.loads(self.rfile.read(length) or b"{}")
+ if not isinstance(value, dict):
+ raise ValueError("JSON request must be an object")
+ return value
+
+ def do_GET(self) -> None:
+ path = urlsplit(self.path).path
+ if path == "/health":
+ self._json({"status": "ok"})
+ return
+ if path in {"/v1/models", "/models"}:
+ self._json({
+ "object": "list",
+ "data": [{
+ "id": gateway.session.runtime.model_id,
+ "object": "model",
+ "created": 0,
+ "owned_by": "planner-cache",
+ }],
+ })
+ return
+ if path == "/props":
+ runtime = gateway.session.runtime
+ self._json({
+ "default_generation_settings": {
+ "params": {
+ "n_predict": runtime.max_new_tokens,
+ "max_tokens": runtime.max_new_tokens,
+ "temperature": runtime.temperature,
+ "top_p": runtime.top_p,
+ "stream": True,
+ }
+ },
+ "total_slots": 1,
+ "model_path": str(runtime.model_path),
+ "chat_template": "Planner Cache runtime-managed chat",
+ "chat_template_caps": {},
+ "modalities": {"vision": False},
+ "build_info": "planner-cache-gateway",
+ "is_sleeping": False,
+ })
+ return
+ if path in {"/slots", "/tools", "/mcp-servers"}:
+ self._json([])
+ return
+ if path == "/planner-cache/state":
+ with gateway.lock:
+ value = gateway.session.state.snapshot()
+ self._json(value)
+ return
+ if path == "/planner-cache/events":
+ with gateway.lock:
+ value = list(gateway.session.recorder.recent_events)
+ self._json(value)
+ return
+ if path == "/planner-cache/personality":
+ try:
+ query = parse_qs(urlsplit(self.path).query)
+ limit = int(query.get("limit", ["100"])[0])
+ offset = int(query.get("offset", ["0"])[0])
+ with gateway.lock:
+ value = gateway.session.command(
+ "/personality", source="web",
+ personality_limit=limit,
+ personality_offset=offset,
+ )
+ except (TypeError, ValueError) as error:
+ self._json({"error": {"message": str(error)}}, 400)
+ return
+ self._json(value)
+ return
+ super().do_GET()
+
+ def do_POST(self) -> None:
+ path = urlsplit(self.path).path
+ if path not in {"/v1/chat/completions", "/chat/completions"}:
+ self._json({"error": {"message": "unsupported gateway endpoint"}}, 404)
+ return
+ try:
+ request = self._read_json()
+ message = newest_user_message(request)
+ messages = native_messages(request)
+ gateway.lock.acquire()
+ try:
+ result = gateway.session.chat(
+ message, raw_messages=messages,
+ )
+ payload = completion_payload(
+ model=gateway.session.runtime.model_id,
+ text=result.text,
+ input_tokens=result.input_tokens,
+ output_tokens=result.output_tokens,
+ )
+ try:
+ if bool(request.get("stream")):
+ self._stream(payload)
+ else:
+ self._json(payload)
+ finally:
+ gateway.session.complete_turn_review()
+ finally:
+ gateway.lock.release()
+ except Exception as error:
+ gateway.session.recorder.event(
+ "ERROR", source="web_gateway",
+ error_type=type(error).__name__, message=str(error),
+ )
+ self._json({"error": {"message": str(error)}}, 500)
+
+ def _stream(self, payload: dict[str, object]) -> None:
+ choice = payload["choices"][0]
+ message = choice["message"]
+ chunk = {
+ "id": payload["id"],
+ "object": "chat.completion.chunk",
+ "created": payload["created"],
+ "model": payload["model"],
+ "choices": [{
+ "index": 0,
+ "delta": {
+ "role": "assistant", "content": message["content"],
+ },
+ "finish_reason": None,
+ }],
+ }
+ finish = {
+ **chunk,
+ "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}],
+ "usage": payload["usage"],
+ }
+ rows = (
+ f"data: {json.dumps(chunk, ensure_ascii=False)}\n\n"
+ f"data: {json.dumps(finish, ensure_ascii=False)}\n\n"
+ "data: [DONE]\n\n"
+ ).encode("utf-8")
+ self.send_response(HTTPStatus.OK)
+ self.send_header("Content-Type", "text/event-stream")
+ self.send_header("Cache-Control", "no-cache")
+ self.send_header("Connection", "close")
+ self.end_headers()
+ self.wfile.write(rows)
+ self.wfile.flush()
+
+ return Handler
+
+ def start(self) -> None:
+ self.thread = threading.Thread(
+ target=self.httpd.serve_forever,
+ name="planner-cache-web-ui",
+ daemon=True,
+ )
+ self.thread.start()
+
+ def close(self) -> None:
+ self.httpd.shutdown()
+ self.httpd.server_close()
+ if self.thread is not None:
+ self.thread.join(timeout=5)
+ self.thread = None
+
+
+def run(args: argparse.Namespace) -> int:
+ session: PlannerChatSession | None = None
+ gateway: PlannerWebGateway | None = None
+ reason = "normal"
+ try:
+ session = PlannerChatSession(args)
+ gateway = PlannerWebGateway(
+ session, args.web_ui_path, host=args.web_host, port=args.web_port,
+ )
+ gateway.start()
+ print(f"\n{session.title}")
+ print(f"Web UI: {gateway.url}")
+ print(f"Session records: {session.recorder.directory}")
+ print("Chat in the browser. Terminal commands: /help /state /personality /events /save /quit\n")
+ if not args.no_browser:
+ webbrowser.open(gateway.url, new=2)
+ while True:
+ try:
+ command = input("Debug: ").strip()
+ except EOFError:
+ # A non-interactive launcher should keep serving until signalled.
+ while True:
+ time.sleep(1)
+ if not command:
+ continue
+ if not command.startswith("/"):
+ print("Chat in the Web UI. Terminal input accepts slash commands only.")
+ continue
+ with gateway.lock:
+ value = session.command(command)
+ if command.casefold() == "/help":
+ value = WEB_HELP
+ if command.casefold() == "/quit":
+ reason = "quit"
+ break
+ if isinstance(value, str):
+ print(value)
+ else:
+ display_json(value)
+ except KeyboardInterrupt:
+ reason = "ctrl-c"
+ print("\nStopping.")
+ finally:
+ if gateway is not None:
+ gateway.close()
+ if session is not None:
+ session.close(reason=reason)
+ return 0
+
+
+def main() -> None:
+ try:
+ raise SystemExit(run(parser().parse_args()))
+ except (FileNotFoundError, ValueError, RuntimeError, OSError) as error:
+ print(f"Planner Cache startup failed: {error}", file=sys.stderr)
+ raise SystemExit(2)
+
+
+if __name__ == "__main__":
+ main()