File size: 13,847 Bytes
1f71c7d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
"""PALIMPSESTE — The autonomous active-inference loop (spec section 4).

The loop, exactly as the spec prescribes (with our bounded implementations of
the open steps 6 & 7)::

    à chaque instant t :
      1. observer o_t
      2. prédir  ô_t = unbind( Φ(bind(self, s_t)), s_t )
      3. surprise ε_t = d(o_t, ô_t)
      4. SI ε_t > seuil τ : écrire (bind(s_t, ctx_t), o_t)  -> apprentissage immédiat
      5. curiosité : choisir l'action a_t qui maximise E[ε_{t+1}]
      6. consolidation (lent) : co-activations fréquemment répétées -> nouveau symbole
      7. méta (très lent) : si une stratégie Φ' réduirait E[ε] -> écrire Φ' dans H_meta

Pas de labels. Pas de superviseur. Pas de reward externe. Le système est mû
uniquement par la minimisation de sa propre surprise.

This module wires together Memory, Phi, Learner, Consolidator, MetaController
into a single ``Palimseste`` agent that runs the loop over a stream of
observations. It is deliberately *environment-agnostic*: the user supplies an
:class:`Environment` adapter (observe/action) and a :class:`StateProjector`
that turns (observation, action) into hypervectors. This keeps the substrate
pure and lets it learn *any* modality.
"""

from __future__ import annotations

from dataclasses import dataclass, field
import math
import numpy as np

from .hv import HV, bind, unbind, bundle, random_hv, similarity, constant_hv
from .memory import Memory, Trace
from .phi import Phi, KernelConfig, Retrieval
from .learner import Learner, Encoder, Prediction
from .consolidation import Consolidator, ConsolidationConfig, ConsolidationResult
from .meta import MetaController, LyapunovEnergy, MetaDecision, max_radius_invariant

__all__ = [
    "Environment",
    "StateProjector",
    "LoopConfig",
    "StepReport",
    "Palimseste",
]


# ----------------------------------------------------------------- interfaces
@dataclass
class Environment:
    """Minimal environment interface (override or subclass).

    The substrate is modality-agnostic; this adapter turns a concrete domain
    into ``(observation, possible_actions)`` pairs the loop can consume.
    """

    def observe(self) -> HV:
        raise NotImplementedError

    def actions(self) -> list[HV]:
        """Return the set of candidate action hypervectors for this tick."""
        raise NotImplementedError

    def act(self, action: HV) -> None:
        raise NotImplementedError

    def done(self) -> bool:
        return False


@dataclass
class StateProjector:
    """Projects (observation, action, history) -> hypervector state ``s_t``.

    The default implementation bundles the last few observations and the
    chosen action into a recurrent state HV. Override ``project`` for custom
    recurrent dynamics.
    """

    D: int
    encoder: Encoder
    window: int = 4
    _history: list[HV] = field(default_factory=list)

    def project(self, obs: HV, action: HV | None) -> HV:
        """Return ``s_t`` from the *past* observations + last action.

        Crucially, ``obs`` is **not** folded into the state yet — the state is
        built from history so that predicting ``obs`` from ``s_t`` is genuine
        prediction, not identity. Call :meth:`commit` after the tick to fold
        ``obs`` into the recurrent history.
        """
        parts: list[HV] = []
        for i, o in enumerate(self._history):
            parts.append(bind(o, self.encoder._role(i)))
        if action is not None:
            parts.append(bind(action, self.encoder._role(self.window)))
        if not parts:
            return random_hv(self.D)
        return bundle(parts, rng=self.encoder.rng)

    def commit(self, obs: HV) -> None:
        """Fold ``obs`` into the recurrent history (call at end of tick)."""
        self._history.append(obs)
        if len(self._history) > self.window:
            self._history = self._history[-self.window:]

    def reset(self) -> None:
        self._history.clear()


# ----------------------------------------------------------------- config
@dataclass
class LoopConfig:
    """Top-level tuning for the autonomous loop.

    Attributes
    ----------
    surprise_threshold : float
        ``τ``. Only write to M when surprise exceeds this (avoids flooding M
        with trivia the system already predicts well).
    curiosity_noise : float
        Std of Gaussian noise added to the curiosity score of each candidate
        action (epsilon-greedy-ish exploration; pure argmax would collapse).
    consolidate_every : int
        Run consolidation (step 6) every N ticks.
    meta_every : int
        Run meta-rewrite (step 7) every N ticks.
    max_radius : int
        Hard invariant cap on the kernel radius (alignment / stability).
    """

    surprise_threshold: float = 0.3
    curiosity_noise: float = 0.05
    consolidate_every: int = 32
    meta_every: int = 128
    max_radius: int = 200

    def __post_init__(self) -> None:
        if not 0.0 <= self.surprise_threshold <= 1.0:
            raise ValueError("surprise_threshold must be in [0, 1]")
        if self.consolidate_every <= 0 or self.meta_every <= 0:
            raise ValueError("periods must be positive")
        if self.max_radius <= 0:
            raise ValueError("max_radius must be positive")


# ----------------------------------------------------------------- reports
@dataclass
class StepReport:
    """Per-tick telemetry for inspection/logging."""

    t: int
    surprise: float
    learned: bool
    action_idx: int | None
    consolidation: ConsolidationResult | None
    meta: MetaDecision | None
    n_traces: int
    n_concepts: int


# ----------------------------------------------------------------- the agent
@dataclass
class Palimseste:
    """The full autonomous PALIMPSESTE agent.

    Composes every subsystem into the active-inference loop. Construction is
    cheap; state lives in ``Memory``.
    """

    D: int = 2000
    loop_cfg: LoopConfig = field(default_factory=LoopConfig)
    rng: np.random.Generator = field(default_factory=np.random.default_rng)
    # subsystems (built lazily in __post_init__)
    mem: Memory | None = None
    phi: Phi | None = None
    learner: Learner | None = None
    encoder: Encoder | None = None
    consolidator: Consolidator | None = None
    meta: MetaController | None = None
    state_projector: StateProjector | None = None
    _t: int = 0
    _last_surprise: float = 0.0

    def __post_init__(self) -> None:
        if self.mem is None:
            self.mem = Memory(D=self.D, rng=self.rng)
        if self.encoder is None:
            self.encoder = Encoder(D=self.D, rng=self.rng)
        if self.phi is None:
            self.phi = Phi(config=KernelConfig(radius=10, min_weight=1e-6))
        if self.learner is None:
            self.learner = Learner(mem=self.mem, phi=self.phi, rng=self.rng)
        if self.consolidator is None:
            self.consolidator = Consolidator(
                mem=self.mem,
                config=ConsolidationConfig(min_pair_sim=-1.0),
                rng=self.rng,
            )
        if self.state_projector is None:
            self.state_projector = StateProjector(
                D=self.D, encoder=self.encoder, window=4
            )
        if self.meta is None:
            energy = LyapunovEnergy(
                invariants=[max_radius_invariant(self.loop_cfg.max_radius)]
            )
            self.meta = MetaController(
                mem=self.mem, phi=self.phi, energy=energy, rng=self.rng
            )

    # ----------------------------------------------------------- one tick
    def step(self, env: Environment) -> StepReport:
        """Run one iteration of the autonomous loop (steps 1-7)."""
        assert self.mem is not None and self.phi is not None
        assert self.learner is not None and self.consolidator is not None
        assert self.meta is not None and self.state_projector is not None
        assert self.encoder is not None

        self._t += 1
        t = self._t

        # 1. observe o_t
        obs = env.observe()

        # build state s_t from *past* observations + last action (does NOT yet
        # include o_t, so predicting o_t from s_t is genuine prediction)
        last_action = getattr(self, "_last_action", None)
        s_t = self.state_projector.project(obs, last_action)

        # 2. predict ô_t = unbind( Phi(bind(self, s_t)), s_t )
        #    The agent learned to map (self, s_t) -> bind(o_t, s_t), so that
        #    unbind(Phi(...), s_t) recovers o_t. See step 4 below.
        self_hv = self._self_hv()
        pred_query = bind(self_hv, s_t)
        pred_val = self.phi(self.mem, pred_query)
        predicted_obs = unbind(pred_val, s_t) if pred_val is not None else None

        # 3. surprise ε_t = d(o_t, ô_t)
        if predicted_obs is None:
            surprise = 1.0
        else:
            surprise = 1.0 - max(0.0, similarity(obs, predicted_obs))
        self._last_surprise = surprise

        # 4. if ε > τ: write (bind(self, s_t), bind(o_t, s_t)) so that
        #    unbind(Phi(bind(self, s_t)), s_t) == o_t on the next similar state.
        #    This matches the spec's predict algebra exactly.
        learned = False
        if surprise > self.loop_cfg.surprise_threshold:
            self.learner.learn(self_hv, bind(obs, s_t), ctx=s_t, weight=1.0,
                               tag=f"obs@t{t}")
            learned = True

        # 5. curiosity: pick the action maximizing E[ε_{t+1}]
        #    We estimate E[ε_{t+1}] per candidate action by predicting what
        #    we'd observe after taking it (query bind(self, bind(s_t, a))).
        actions = env.actions()
        action_idx = self._curious_action(actions, s_t, self_hv)
        if action_idx is not None:
            chosen = actions[action_idx]
            env.act(chosen)
            self._last_action = chosen
        else:
            self._last_action = None

        # fold the current observation into recurrent history for next tick
        self.state_projector.commit(obs)

        # 6. consolidation (slow): every N ticks
        cons_result: ConsolidationResult | None = None
        if t % self.loop_cfg.consolidate_every == 0 and self.mem.traces:
            # observe co-activations from the latest retrieval neighborhood
            ret = self.phi.retrieve(self.mem, pred_query)
            self.consolidator.observe_retrieval([tr.id for tr in ret.matches])
            cons_result = self.consolidator.consolidate()

        # 7. meta (very slow): every M ticks
        meta_result: MetaDecision | None = None
        if t % self.loop_cfg.meta_every == 0 and len(self.mem.traces) >= 8:
            replay = self.meta.build_replay(32)
            if replay:
                meta_result = self.meta.step(replay, max_proposals=8)

        return StepReport(
            t=t,
            surprise=surprise,
            learned=learned,
            action_idx=action_idx,
            consolidation=cons_result,
            meta=meta_result,
            n_traces=len(self.mem),
            n_concepts=self.consolidator.n_promoted,
        )

    # ----------------------------------------------------------- curiosity
    def _curious_action(self, actions: list[HV], s_t: HV, self_hv: HV) -> int | None:
        """Pick the action that maximizes *expected* next surprise.

        ``E[ε_{t+1}]`` is estimated by: for each candidate action ``a``,
        predict the next observation ``ô' = Phi(bind(self, bind(s_t, a)))``,
        and score ``a`` by ``1 - confidence(ô')`` (low confidence = high
        expected surprise = high curiosity). Noise is added to avoid collapse.
        """
        assert self.phi is not None and self.mem is not None
        if not actions:
            return None
        if len(actions) == 1:
            return 0
        scores = np.empty(len(actions))
        for i, a in enumerate(actions):
            q = bind(self_hv, bind(s_t, a))
            ret = self.phi.retrieve(self.mem, q)
            if ret.matches:
                ws = np.asarray(ret.weights)
                sims = np.asarray(ret.sims)
                conf = float(np.clip((ws * sims).sum() / ws.sum(), -1.0, 1.0))
                conf = (conf + 1.0) / 2.0
            else:
                conf = 0.0
            # curiosity = 1 - confidence (we want to explore what we can't predict)
            scores[i] = 1.0 - conf + self.rng.normal(0.0, self.loop_cfg.curiosity_noise)
        return int(np.argmax(scores))

    # ----------------------------------------------------------- self-identity
    def _self_hv(self) -> HV:
        """A stable identity HV for the agent (the 'self' in the spec).

        Lazily created and cached on the instance.
        """
        cached = getattr(self, "_self_hv_cache", None)
        if cached is None:
            cached = random_hv(self.D, rng=self.rng)
            self._self_hv_cache = cached
        return cached

    # ----------------------------------------------------------- utilities
    def reset_state(self) -> None:
        """Clear recurrent state (e.g. between episodes). Does NOT clear M."""
        assert self.state_projector is not None
        self.state_projector.reset()
        self._last_action = None
        self._last_surprise = 0.0

    @property
    def surprise(self) -> float:
        return self._last_surprise

    def stats(self) -> dict:
        assert self.mem is not None and self.consolidator is not None and self.meta is not None
        s = self.mem.stats()
        return {
            "t": self._t,
            "n_traces": s.n_traces,
            "n_meta": s.n_meta,
            "n_concepts": self.consolidator.n_promoted,
            "n_meta_decisions": len(self.meta.history),
            "last_surprise": self._last_surprise,
            "kernel": self.meta.config.encode(),
        }