File size: 13,847 Bytes
1f71c7d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 | """PALIMPSESTE — The autonomous active-inference loop (spec section 4).
The loop, exactly as the spec prescribes (with our bounded implementations of
the open steps 6 & 7)::
à chaque instant t :
1. observer o_t
2. prédir ô_t = unbind( Φ(bind(self, s_t)), s_t )
3. surprise ε_t = d(o_t, ô_t)
4. SI ε_t > seuil τ : écrire (bind(s_t, ctx_t), o_t) -> apprentissage immédiat
5. curiosité : choisir l'action a_t qui maximise E[ε_{t+1}]
6. consolidation (lent) : co-activations fréquemment répétées -> nouveau symbole
7. méta (très lent) : si une stratégie Φ' réduirait E[ε] -> écrire Φ' dans H_meta
Pas de labels. Pas de superviseur. Pas de reward externe. Le système est mû
uniquement par la minimisation de sa propre surprise.
This module wires together Memory, Phi, Learner, Consolidator, MetaController
into a single ``Palimseste`` agent that runs the loop over a stream of
observations. It is deliberately *environment-agnostic*: the user supplies an
:class:`Environment` adapter (observe/action) and a :class:`StateProjector`
that turns (observation, action) into hypervectors. This keeps the substrate
pure and lets it learn *any* modality.
"""
from __future__ import annotations
from dataclasses import dataclass, field
import math
import numpy as np
from .hv import HV, bind, unbind, bundle, random_hv, similarity, constant_hv
from .memory import Memory, Trace
from .phi import Phi, KernelConfig, Retrieval
from .learner import Learner, Encoder, Prediction
from .consolidation import Consolidator, ConsolidationConfig, ConsolidationResult
from .meta import MetaController, LyapunovEnergy, MetaDecision, max_radius_invariant
__all__ = [
"Environment",
"StateProjector",
"LoopConfig",
"StepReport",
"Palimseste",
]
# ----------------------------------------------------------------- interfaces
@dataclass
class Environment:
"""Minimal environment interface (override or subclass).
The substrate is modality-agnostic; this adapter turns a concrete domain
into ``(observation, possible_actions)`` pairs the loop can consume.
"""
def observe(self) -> HV:
raise NotImplementedError
def actions(self) -> list[HV]:
"""Return the set of candidate action hypervectors for this tick."""
raise NotImplementedError
def act(self, action: HV) -> None:
raise NotImplementedError
def done(self) -> bool:
return False
@dataclass
class StateProjector:
"""Projects (observation, action, history) -> hypervector state ``s_t``.
The default implementation bundles the last few observations and the
chosen action into a recurrent state HV. Override ``project`` for custom
recurrent dynamics.
"""
D: int
encoder: Encoder
window: int = 4
_history: list[HV] = field(default_factory=list)
def project(self, obs: HV, action: HV | None) -> HV:
"""Return ``s_t`` from the *past* observations + last action.
Crucially, ``obs`` is **not** folded into the state yet — the state is
built from history so that predicting ``obs`` from ``s_t`` is genuine
prediction, not identity. Call :meth:`commit` after the tick to fold
``obs`` into the recurrent history.
"""
parts: list[HV] = []
for i, o in enumerate(self._history):
parts.append(bind(o, self.encoder._role(i)))
if action is not None:
parts.append(bind(action, self.encoder._role(self.window)))
if not parts:
return random_hv(self.D)
return bundle(parts, rng=self.encoder.rng)
def commit(self, obs: HV) -> None:
"""Fold ``obs`` into the recurrent history (call at end of tick)."""
self._history.append(obs)
if len(self._history) > self.window:
self._history = self._history[-self.window:]
def reset(self) -> None:
self._history.clear()
# ----------------------------------------------------------------- config
@dataclass
class LoopConfig:
"""Top-level tuning for the autonomous loop.
Attributes
----------
surprise_threshold : float
``τ``. Only write to M when surprise exceeds this (avoids flooding M
with trivia the system already predicts well).
curiosity_noise : float
Std of Gaussian noise added to the curiosity score of each candidate
action (epsilon-greedy-ish exploration; pure argmax would collapse).
consolidate_every : int
Run consolidation (step 6) every N ticks.
meta_every : int
Run meta-rewrite (step 7) every N ticks.
max_radius : int
Hard invariant cap on the kernel radius (alignment / stability).
"""
surprise_threshold: float = 0.3
curiosity_noise: float = 0.05
consolidate_every: int = 32
meta_every: int = 128
max_radius: int = 200
def __post_init__(self) -> None:
if not 0.0 <= self.surprise_threshold <= 1.0:
raise ValueError("surprise_threshold must be in [0, 1]")
if self.consolidate_every <= 0 or self.meta_every <= 0:
raise ValueError("periods must be positive")
if self.max_radius <= 0:
raise ValueError("max_radius must be positive")
# ----------------------------------------------------------------- reports
@dataclass
class StepReport:
"""Per-tick telemetry for inspection/logging."""
t: int
surprise: float
learned: bool
action_idx: int | None
consolidation: ConsolidationResult | None
meta: MetaDecision | None
n_traces: int
n_concepts: int
# ----------------------------------------------------------------- the agent
@dataclass
class Palimseste:
"""The full autonomous PALIMPSESTE agent.
Composes every subsystem into the active-inference loop. Construction is
cheap; state lives in ``Memory``.
"""
D: int = 2000
loop_cfg: LoopConfig = field(default_factory=LoopConfig)
rng: np.random.Generator = field(default_factory=np.random.default_rng)
# subsystems (built lazily in __post_init__)
mem: Memory | None = None
phi: Phi | None = None
learner: Learner | None = None
encoder: Encoder | None = None
consolidator: Consolidator | None = None
meta: MetaController | None = None
state_projector: StateProjector | None = None
_t: int = 0
_last_surprise: float = 0.0
def __post_init__(self) -> None:
if self.mem is None:
self.mem = Memory(D=self.D, rng=self.rng)
if self.encoder is None:
self.encoder = Encoder(D=self.D, rng=self.rng)
if self.phi is None:
self.phi = Phi(config=KernelConfig(radius=10, min_weight=1e-6))
if self.learner is None:
self.learner = Learner(mem=self.mem, phi=self.phi, rng=self.rng)
if self.consolidator is None:
self.consolidator = Consolidator(
mem=self.mem,
config=ConsolidationConfig(min_pair_sim=-1.0),
rng=self.rng,
)
if self.state_projector is None:
self.state_projector = StateProjector(
D=self.D, encoder=self.encoder, window=4
)
if self.meta is None:
energy = LyapunovEnergy(
invariants=[max_radius_invariant(self.loop_cfg.max_radius)]
)
self.meta = MetaController(
mem=self.mem, phi=self.phi, energy=energy, rng=self.rng
)
# ----------------------------------------------------------- one tick
def step(self, env: Environment) -> StepReport:
"""Run one iteration of the autonomous loop (steps 1-7)."""
assert self.mem is not None and self.phi is not None
assert self.learner is not None and self.consolidator is not None
assert self.meta is not None and self.state_projector is not None
assert self.encoder is not None
self._t += 1
t = self._t
# 1. observe o_t
obs = env.observe()
# build state s_t from *past* observations + last action (does NOT yet
# include o_t, so predicting o_t from s_t is genuine prediction)
last_action = getattr(self, "_last_action", None)
s_t = self.state_projector.project(obs, last_action)
# 2. predict ô_t = unbind( Phi(bind(self, s_t)), s_t )
# The agent learned to map (self, s_t) -> bind(o_t, s_t), so that
# unbind(Phi(...), s_t) recovers o_t. See step 4 below.
self_hv = self._self_hv()
pred_query = bind(self_hv, s_t)
pred_val = self.phi(self.mem, pred_query)
predicted_obs = unbind(pred_val, s_t) if pred_val is not None else None
# 3. surprise ε_t = d(o_t, ô_t)
if predicted_obs is None:
surprise = 1.0
else:
surprise = 1.0 - max(0.0, similarity(obs, predicted_obs))
self._last_surprise = surprise
# 4. if ε > τ: write (bind(self, s_t), bind(o_t, s_t)) so that
# unbind(Phi(bind(self, s_t)), s_t) == o_t on the next similar state.
# This matches the spec's predict algebra exactly.
learned = False
if surprise > self.loop_cfg.surprise_threshold:
self.learner.learn(self_hv, bind(obs, s_t), ctx=s_t, weight=1.0,
tag=f"obs@t{t}")
learned = True
# 5. curiosity: pick the action maximizing E[ε_{t+1}]
# We estimate E[ε_{t+1}] per candidate action by predicting what
# we'd observe after taking it (query bind(self, bind(s_t, a))).
actions = env.actions()
action_idx = self._curious_action(actions, s_t, self_hv)
if action_idx is not None:
chosen = actions[action_idx]
env.act(chosen)
self._last_action = chosen
else:
self._last_action = None
# fold the current observation into recurrent history for next tick
self.state_projector.commit(obs)
# 6. consolidation (slow): every N ticks
cons_result: ConsolidationResult | None = None
if t % self.loop_cfg.consolidate_every == 0 and self.mem.traces:
# observe co-activations from the latest retrieval neighborhood
ret = self.phi.retrieve(self.mem, pred_query)
self.consolidator.observe_retrieval([tr.id for tr in ret.matches])
cons_result = self.consolidator.consolidate()
# 7. meta (very slow): every M ticks
meta_result: MetaDecision | None = None
if t % self.loop_cfg.meta_every == 0 and len(self.mem.traces) >= 8:
replay = self.meta.build_replay(32)
if replay:
meta_result = self.meta.step(replay, max_proposals=8)
return StepReport(
t=t,
surprise=surprise,
learned=learned,
action_idx=action_idx,
consolidation=cons_result,
meta=meta_result,
n_traces=len(self.mem),
n_concepts=self.consolidator.n_promoted,
)
# ----------------------------------------------------------- curiosity
def _curious_action(self, actions: list[HV], s_t: HV, self_hv: HV) -> int | None:
"""Pick the action that maximizes *expected* next surprise.
``E[ε_{t+1}]`` is estimated by: for each candidate action ``a``,
predict the next observation ``ô' = Phi(bind(self, bind(s_t, a)))``,
and score ``a`` by ``1 - confidence(ô')`` (low confidence = high
expected surprise = high curiosity). Noise is added to avoid collapse.
"""
assert self.phi is not None and self.mem is not None
if not actions:
return None
if len(actions) == 1:
return 0
scores = np.empty(len(actions))
for i, a in enumerate(actions):
q = bind(self_hv, bind(s_t, a))
ret = self.phi.retrieve(self.mem, q)
if ret.matches:
ws = np.asarray(ret.weights)
sims = np.asarray(ret.sims)
conf = float(np.clip((ws * sims).sum() / ws.sum(), -1.0, 1.0))
conf = (conf + 1.0) / 2.0
else:
conf = 0.0
# curiosity = 1 - confidence (we want to explore what we can't predict)
scores[i] = 1.0 - conf + self.rng.normal(0.0, self.loop_cfg.curiosity_noise)
return int(np.argmax(scores))
# ----------------------------------------------------------- self-identity
def _self_hv(self) -> HV:
"""A stable identity HV for the agent (the 'self' in the spec).
Lazily created and cached on the instance.
"""
cached = getattr(self, "_self_hv_cache", None)
if cached is None:
cached = random_hv(self.D, rng=self.rng)
self._self_hv_cache = cached
return cached
# ----------------------------------------------------------- utilities
def reset_state(self) -> None:
"""Clear recurrent state (e.g. between episodes). Does NOT clear M."""
assert self.state_projector is not None
self.state_projector.reset()
self._last_action = None
self._last_surprise = 0.0
@property
def surprise(self) -> float:
return self._last_surprise
def stats(self) -> dict:
assert self.mem is not None and self.consolidator is not None and self.meta is not None
s = self.mem.stats()
return {
"t": self._t,
"n_traces": s.n_traces,
"n_meta": s.n_meta,
"n_concepts": self.consolidator.n_promoted,
"n_meta_decisions": len(self.meta.history),
"last_surprise": self._last_surprise,
"kernel": self.meta.config.encode(),
}
|