pareidolia / mind /backends.py
AndresCarreon's picture
pareidolia: post-snap geometry gate (one multi-turn feature retry) + eye readability halo + wall/img/SSE container-nap resilience
de59038 verified
Raw
History Blame Contribute Delete
49.9 kB
"""Spirit-medium backends: MockMedium (canned awakenings) and ZeroGPUMedium.
Selected via the PAREIDOLIA_BACKEND env: mock (default) | zerogpu.
- MockMedium: deterministic canned awakening records keyed by image hash,
with a realistic configurable delay — the whole frontend, server, and test
suite build against this with zero GPU and zero ML imports.
- ZeroGPUMedium: MiniCPM-V-4_5 per ARCHITECTURE.md §7. ALL ML imports are
confined to its load path, so importing this module in mock mode never
touches torch.
This module also owns the combined-GPU-window plumbing (§2): `awaken_full`
runs VLM -> CV snap -> geometry retry -> grudge punch-up -> grudge TTS ->
mutter TTS inside ONE
@spaces.GPU(duration=75) function so a visitor pays one queue wait and one
quota spend. The mutter TTS (same persona voice) is best-effort polish gated
by MUTTER_MIN_MARGIN_SECONDS of remaining window margin — it can be skipped
or fail without ever costing the awakening. The geometry retry (live medium
only) re-asks ONCE for different face features when the post-snap geometry
fails (schema.face_geometry_violations); it is skipped when the repair retry
is already spent or under GEOMETRY_MIN_MARGIN_SECONDS of margin remain, and
the better-scoring feature set ships either way. The punch-up is a second
short SAMPLED generation (live medium only) that rewrites the grudge line; it
applies only under the strict acceptance bar in prompts.punchup_rejection and
is skipped when under PUNCHUP_MIN_MARGIN_SECONDS of window margin remain. `spaces` is imported (and
the decorator applied) at module import time when available — ZeroGPU
discovers GPU functions at startup, and importing `spaces` here also
guarantees it precedes any torch import (§7 hard rule). Locally the bare
function runs instead.
"""
from __future__ import annotations
import io
import logging
import os
import re
import threading
import time
from typing import Any, Callable, Optional
from .schema import (
AwakeningParseError,
AwakeningResult,
PoeticError,
build_geometry_repair_prompt,
build_repair_prompt,
face_geometry_violations,
parse_awakening,
)
from .voice import make_voice
logger = logging.getLogger("pareidolia.mind")
# ---------------------------------------------------------------------------
# Séance prompt (writing agent's mind/prompts.py owns the real one via
# build_seance_prompt()). Import-guarded with a minimal-but-correct fallback
# so the zerogpu path can bench even without prompts.py.
# ---------------------------------------------------------------------------
try:
from .prompts import build_seance_prompt as _build_seance_prompt
SEANCE_PROMPT: str = _build_seance_prompt()
except Exception: # noqa: BLE001 - any import problem means "use the fallback"
SEANCE_PROMPT = (
"You are a spirit medium. This object has a latent face made of its"
" EXISTING visual features, and a soul shaped by its visible condition."
" Find both. First write a one-line critique of your feature choices,"
" then commit. Reply with ONLY one JSON object, no code fences:\n"
"{\n"
' "gate": {"contains_human_face": bool, "nsfw": bool,'
' "recognizable_object": bool},\n'
' "object": str, "material": str, "condition": str, "setting": str,\n'
' "candidate_features": [{"name": str, "role":'
' "eye_left"|"eye_right"|"mouth", "cx": 0..1, "cy": 0..1,'
' "size": 0..1}],\n'
' "critique": str,\n'
' "persona": {"archetype": str, "voice": str, "mood": str},\n'
' "lines": {"grudge": str, "mutter": str}\n'
"}\n"
"Exactly one eye_left, one eye_right, at most one mouth — each an"
" existing visual element, named as seen. Every line must reference a"
" visible specific of THIS object. Dry, deadpan, dignified; <= 22"
" words per line; PG-13; never open with 'I am'."
)
# ---------------------------------------------------------------------------
# Punch-up brief + acceptance bar (prompts.py owns both; pure functions, no
# ML). Import-guarded like the séance prompt so the zerogpu path can bench
# even without prompts.py — the punch-up then simply never applies and the
# original grudge ships.
# ---------------------------------------------------------------------------
try:
from .prompts import build_punchup_prompt as _build_punchup_prompt
from .prompts import punchup_rejection as _punchup_rejection
except Exception: # noqa: BLE001 - any import problem means "no punch-up"
_build_punchup_prompt = None # type: ignore[assignment]
_punchup_rejection = None # type: ignore[assignment]
# ---------------------------------------------------------------------------
# CV snapping (cv agent's cv/snap.py). Assumed contract:
# snap_features(image, features: list[dict]) -> list[dict]
# where each returned feature keeps name/role/cx/cy/size and gains
# "snap_delta" (normalized distance moved). Identity fallback keeps the
# VLM's coarse points — the mist forgives ±15% by design (§0).
# ---------------------------------------------------------------------------
try: # pragma: no cover - exercised only once cv/snap.py lands
from cv.snap import snap_features as _snap_features # type: ignore[import-not-found]
except Exception: # noqa: BLE001
def _snap_features(image: Any, features: list[dict]) -> list[dict]:
"""Identity fallback: keep coarse points, report zero snap delta."""
return [{**feature, "snap_delta": 0.0} for feature in features]
def _as_snap_image(image: Any) -> Any:
"""Best-effort convert to the BGR uint8 ndarray cv/snap.py wants.
The transport hands the pipeline a PIL image; cv.snap.snap_features speaks
ndarray only (its documented contract). Anything without ``.convert``
(bytes in tests, an ndarray already) passes through unchanged — the snap
try/except in :func:`_pipeline` keeps the VLM's coarse points whenever the
shape is wrong, and the mist forgives ±15% (§0).
"""
convert = getattr(image, "convert", None)
if convert is None:
return image
import numpy as np # transitively present via opencv; local to stay light
rgb = np.asarray(convert("RGB"))
return np.ascontiguousarray(rgb[:, :, ::-1]) # RGB -> BGR
# ---------------------------------------------------------------------------
# Canned awakening records (mock backend). HYDRANT_RECORD is EXACTLY the §3
# fire-hydrant example and the default for unknown images. The eight
# _M*_RECORDs are COHERENT souls for the eight dev photos in
# web/mock/photos/ (same objects, coordinates eyeballed against the actual
# pixels; lines reuse/match web/mock/records.json, written to the WRITING.md
# bar) — recognized by perceptual average-hash so a mock demo never captions
# a hydrant photo as "the weathered traffic cone".
# ---------------------------------------------------------------------------
HYDRANT_RECORD: dict = {
"gate": {"contains_human_face": False, "nsfw": False, "recognizable_object": True},
"object": "fire hydrant",
"material": "cast iron",
"condition": "rusted",
"setting": "sidewalk, residential street",
"candidate_features": [
{"name": "left bonnet bolt", "role": "eye_left", "cx": 0.42, "cy": 0.31, "size": 0.06},
{"name": "right bonnet bolt", "role": "eye_right", "cx": 0.58, "cy": 0.31, "size": 0.06},
{"name": "front outlet cap", "role": "mouth", "cx": 0.50, "cy": 0.55, "size": 0.12},
],
"critique": "bolts are symmetric and round — strong eyes; outlet sits low-center, good mouth",
"persona": {"archetype": "the_veteran", "voice": "gravel_low", "mood": "long-suffering"},
"lines": {
"grudge": "Forty years on this corner. Not one dog has shown me respect.",
"mutter": "Paint me red, they said. It'll be dignified, they said.",
},
}
_M1_HYDRANT_RECORD: dict = {
"gate": {"contains_human_face": False, "nsfw": False, "recognizable_object": True},
"object": "fire hydrant",
"material": "cast iron",
"condition": "rusted",
"setting": "against a corrugated metal wall",
"candidate_features": [
{"name": "left outlet cap", "role": "eye_left", "cx": 0.337, "cy": 0.328, "size": 0.16},
{"name": "right outlet cap", "role": "eye_right", "cx": 0.547, "cy": 0.33, "size": 0.16},
{"name": "paired flange bolts", "role": "mouth", "cx": 0.415, "cy": 0.49, "size": 0.10},
],
"critique": (
"the two outlet caps sit level and round — strong eyes; the paired "
"flange bolts below read as a small, set mouth"
),
"persona": {"archetype": "the_veteran", "voice": "gravel_low", "mood": "long-suffering"},
"lines": {
"grudge": (
"They painted over the rust twice and never once turned my "
"valve. Decorative, apparently."
),
"mutter": "Two coats of orange. Still thirsty.",
},
}
_M2_CONE_RECORD: dict = {
"gate": {"contains_human_face": False, "nsfw": False, "recognizable_object": True},
"object": "traffic cone",
"material": "weathered pvc",
"condition": "worn",
"setting": "curb edge, residential street",
"candidate_features": [
{"name": "left scuff mark", "role": "eye_left", "cx": 0.40, "cy": 0.385, "size": 0.05},
{"name": "right sun-fade patch", "role": "eye_right", "cx": 0.52, "cy": 0.39, "size": 0.05},
{"name": "moulding seam shadow", "role": "mouth", "cx": 0.46, "cy": 0.56, "size": 0.09},
],
"critique": (
"the scuff and the sun-fade patch sit nearly level — tired but "
"honest eyes; the moulding seam shadow makes a thin, grim mouth"
),
"persona": {"archetype": "the_veteran", "voice": "gravel_low", "mood": "resolute"},
"lines": {
"grudge": (
"Temporary placement, they said. The grass has since eaten the "
"curb. I hold the line alone."
),
"mutter": "Still here. Still 'temporary.'",
},
}
_M3_MUG_RECORD: dict = {
"gate": {"contains_human_face": False, "nsfw": False, "recognizable_object": True},
"object": "coffee mug",
"material": "glazed ceramic",
"condition": "loved",
"setting": "on a fire grate, campfire",
"candidate_features": [
{"name": "ember reflection", "role": "eye_left", "cx": 0.625, "cy": 0.62, "size": 0.05},
{"name": "glaze highlight", "role": "eye_right", "cx": 0.76, "cy": 0.63, "size": 0.05},
{"name": "base shadow curve", "role": "mouth", "cx": 0.69, "cy": 0.78, "size": 0.09},
],
"critique": (
"the ember reflection and the glaze highlight sit level on the "
"curve — lit, wary eyes; the base shadow bends like a resigned mouth"
),
"persona": {"archetype": "the_martyr", "voice": "weary_warm", "mood": "resigned"},
"lines": {
"grudge": (
"They set me on the grill. Directly over the fire. I'm told "
"this is 'camping.'"
),
"mutter": "Nobody checks the handle temperature.",
},
}
_M4_TOASTER_RECORD: dict = {
"gate": {"contains_human_face": False, "nsfw": False, "recognizable_object": True},
"object": "toaster",
"material": "white enamel",
"condition": "pristine",
"setting": "bare counter, studio light",
"candidate_features": [
{"name": "left slot guard", "role": "eye_left", "cx": 0.30, "cy": 0.615, "size": 0.10},
{"name": "right slot guard", "role": "eye_right", "cx": 0.655, "cy": 0.605, "size": 0.10},
{"name": "base seam shadow", "role": "mouth", "cx": 0.50, "cy": 0.84, "size": 0.14},
],
"critique": (
"the slot guards are symmetric and wide-set — immaculate eyes; the "
"base seam shadow runs flat and composed, a professional's mouth"
),
"persona": {"archetype": "the_perfectionist", "voice": "prim_clipped", "mood": "wounded pride"},
"lines": {
"grudge": (
"Two slots. They use one. The other waits, pristine, for "
"guests who never toast."
),
"mutter": "Bagel setting. Untouched since purchase.",
},
}
_M5_MAILBOX_RECORD: dict = {
"gate": {"contains_human_face": False, "nsfw": False, "recognizable_object": True},
"object": "mailbox",
"material": "galvanized steel",
"condition": "weathered",
"setting": "snowdrift, farm fence line, 1940",
"candidate_features": [
# The rivets really are ~0.04 apart on the small door — distinct,
# eyeballed on the photo. The 0.05-separation rule lives in the
# parse_awakening path (VLM degeneracy guard), not the schema.
{"name": "left door rivet", "role": "eye_left", "cx": 0.705, "cy": 0.287, "size": 0.035},
{"name": "right door rivet", "role": "eye_right", "cx": 0.745, "cy": 0.286, "size": 0.035},
{"name": "door latch", "role": "mouth", "cx": 0.728, "cy": 0.363, "size": 0.05},
],
"critique": (
"the two door rivets sit close and level — small wary eyes; the "
"latch beneath makes a tight deadpan mouth"
),
"persona": {"archetype": "the_rival", "voice": "deadpan_flat", "mood": "clipped"},
"lines": {
"grudge": (
"We've stood in this snow since January. R.C. Lenhart gets a "
"name plate. I get 'the other box.'"
),
"mutter": "Her flag works. Allegedly.",
},
}
_M6_POSTBOX_RECORD: dict = {
"gate": {"contains_human_face": False, "nsfw": False, "recognizable_object": True},
"object": "post box",
"material": "painted steel",
"condition": "kept",
"setting": "lawn beside a parking lot",
"candidate_features": [
{"name": "left flap hinge", "role": "eye_left", "cx": 0.345, "cy": 0.295, "size": 0.045},
{"name": "right flap hinge", "role": "eye_right", "cx": 0.60, "cy": 0.295, "size": 0.045},
{"name": "collection keyhole", "role": "mouth", "cx": 0.475, "cy": 0.415, "size": 0.05},
],
"critique": (
"the flap hinges sit wide and perfectly level — patient eyes; the "
"collection keyhole is a small, hopeful mouth"
),
"persona": {"archetype": "the_romantic", "voice": "soft_wistful", "mood": "hopeful"},
"lines": {
"grudge": (
"Parked cars come and go all day. Nobody writes. I keep the "
"slot warm anyway."
),
"mutter": "The slot stays open. Just in case.",
},
}
_M7_ASPIDISTRA_RECORD: dict = {
"gate": {"contains_human_face": False, "nsfw": False, "recognizable_object": True},
"object": "aspidistra",
"material": "leaf and glazed bowl",
"condition": "antique",
"setting": "parlor table, framed pictures behind",
"candidate_features": [
{"name": "left glaze highlight", "role": "eye_left", "cx": 0.425, "cy": 0.755, "size": 0.04},
{"name": "right glaze highlight", "role": "eye_right", "cx": 0.525, "cy": 0.76, "size": 0.04},
{"name": "doily shadow", "role": "mouth", "cx": 0.475, "cy": 0.875, "size": 0.08},
],
"critique": (
"the paired glaze highlights sit level on the bowl — calm, ancient "
"eyes; the doily shadow curves into a serene mouth"
),
"persona": {"archetype": "the_philosopher", "voice": "slow_grand", "mood": "serene"},
"lines": {
"grudge": (
"I outlasted everyone in those three frames. The lesson here is "
"patience. Also, shade tolerance."
),
"mutter": "The frames went first. Noted.",
},
}
_M8_STAPLER_RECORD: dict = {
"gate": {"contains_human_face": False, "nsfw": False, "recognizable_object": True},
"object": "stapler",
"material": "cast metal",
"condition": "stored",
"setting": "bare shelf, yellowed wall",
"candidate_features": [
{"name": "side rivet", "role": "eye_left", "cx": 0.245, "cy": 0.665, "size": 0.03},
{"name": "anvil latch", "role": "eye_right", "cx": 0.455, "cy": 0.615, "size": 0.03},
{"name": "loaded staple strip", "role": "mouth", "cx": 0.545, "cy": 0.745, "size": 0.07},
],
"critique": (
"the side rivet and the anvil latch hold a sidelong gaze — uneven "
"but alive; the loaded staple strip is a clenched mouth"
),
"persona": {"archetype": "the_conspiracist", "voice": "paranoid_whisper", "mood": "vigilant"},
"lines": {
"grudge": (
"The gray one appeared the week the files went missing. Nobody "
"else finds that interesting."
),
"mutter": "It hasn't stapled once. Not once.",
},
}
CANNED_RECORDS: tuple[dict, ...] = (
HYDRANT_RECORD,
_M1_HYDRANT_RECORD,
_M2_CONE_RECORD,
_M3_MUG_RECORD,
_M4_TOASTER_RECORD,
_M5_MAILBOX_RECORD,
_M6_POSTBOX_RECORD,
_M7_ASPIDISTRA_RECORD,
_M8_STAPLER_RECORD,
)
# 64-bit average-hashes of web/mock/photos/m1..m8.jpg (computed June 12 with
# _average_hash below). aHash survives the client's canvas downscale/JPEG
# re-encode and app.py's thumbnail: measured drift ≤1 bit at 0.8x + q70,
# while the closest photo PAIR is 18 bits apart — a ≤8-bit match is
# unambiguous. Same photo -> same soul, forever; anything else -> hydrant.
_PHOTO_HASH_RECORDS: tuple[tuple[int, dict], ...] = (
(0xFFFFFFCED8101010, _M1_HYDRANT_RECORD),
(0xFFF9101818181800, _M2_CONE_RECORD),
(0xF838363F7F701000, _M3_MUG_RECORD),
(0x7E7E7E64003E3C3C, _M4_TOASTER_RECORD),
(0xFEFF31E1E12124F8, _M5_MAILBOX_RECORD),
(0xC0D04003C3FFFCFC, _M6_POSTBOX_RECORD),
(0x0000007F773F79FF, _M7_ASPIDISTRA_RECORD),
(0xFFFFE3E3C18083FF, _M8_STAPLER_RECORD),
)
_PHOTO_HASH_MAX_DISTANCE = 8
def _average_hash(image: Any) -> Optional[int]:
"""64-bit perceptual average-hash, or None when ``image`` isn't one.
Accepts a PIL-like image (has ``.convert``) or raw encoded bytes; strings
and everything else return None (the mock must never read paths or raise).
PIL is a light image-I/O dependency (already required by app.py), NOT an
ML import — the zero-ML mock contract holds.
"""
try:
from PIL import Image # noqa: PLC0415 - light, lazy by convention
if isinstance(image, (bytes, bytearray)):
img = Image.open(io.BytesIO(bytes(image)))
img.load()
elif hasattr(image, "convert"):
img = image
else:
return None
gray = img.convert("L").resize((8, 8), Image.Resampling.LANCZOS)
pixels = gray.tobytes()
avg = sum(pixels) / 64.0
bits = 0
for p in pixels:
bits = (bits << 1) | (1 if p > avg else 0)
return bits
except Exception: # noqa: BLE001 - undecodable bytes, broken PIL object …
return None
def _canned_record_for(image: Any) -> dict:
"""The coherent canned record for a known dev photo, else the §3 hydrant."""
ahash = _average_hash(image)
if ahash is None:
return HYDRANT_RECORD
best: Optional[dict] = None
best_d = _PHOTO_HASH_MAX_DISTANCE + 1
for ref, record in _PHOTO_HASH_RECORDS:
d = (ahash ^ ref).bit_count()
if d < best_d:
best, best_d = record, d
return best if best is not None else HYDRANT_RECORD
class MockMedium:
"""Deterministic canned medium — what every demo and test runs on.
Same image -> same record, forever (the Menagerie replays identically;
judges can refresh without surprises). Each of the eight dev photos in
web/mock/photos/ is recognized by perceptual average-hash and gets its
own COHERENT record (the photo's actual object, features eyeballed on
its pixels) — a mock demo can never caption a hydrant photo as a cone.
Unknown images (and None) get the §3 fire-hydrant example verbatim.
The sleep (default 6 s, PAREIDOLIA_MOCK_DELAY env, constructor override
for tests) keeps the séance animation honest: latency theater must be
rehearsed against realistic latency.
"""
name = "mock"
model_id = "pareidolia-mock-canned"
def __init__(self, delay: Optional[float] = None):
if delay is None:
try:
delay = float(os.environ.get("PAREIDOLIA_MOCK_DELAY", "6"))
except ValueError:
delay = 6.0
self.delay = max(0.0, delay)
def awaken(self, image: Any, prompt: Optional[str] = None) -> AwakeningResult:
"""Return the canned awakening for this image after the mock delay.
`prompt` is accepted (and ignored) so call sites never special-case
the backend. Records round-trip through pydantic so the mock can
never drift from the schema contract.
"""
if self.delay:
time.sleep(self.delay)
record = _canned_record_for(image)
return AwakeningResult.model_validate(record)
# ---------------------------------------------------------------------------
# ZeroGPU medium — MiniCPM-V-4_5
# ---------------------------------------------------------------------------
REFUSAL_SCATTERED = (
"The spirits spoke, but their words scattered like mist. Try another angle."
)
# --- Punch-up window guard ---------------------------------------------------
# The whole awakening runs inside ONE @spaces.GPU(duration=75) window. The
# punch-up is optional polish: it is only attempted while a conservative
# margin remains for itself + TTS; otherwise it is skipped and the original
# grudge ships. p50 is ~19 s of GPU at the punch-up point, so the guard is a
# cheap monotonic check, not a scheduler.
GPU_WINDOW_SECONDS = 75.0
PUNCHUP_MIN_MARGIN_SECONDS = 20.0
def punchup_window_open(elapsed_seconds: float) -> bool:
"""True while >= PUNCHUP_MIN_MARGIN_SECONDS of the GPU window remain."""
return (GPU_WINDOW_SECONDS - elapsed_seconds) >= PUNCHUP_MIN_MARGIN_SECONDS
# --- Mutter window guard -----------------------------------------------------
# The wall's idle mutters become AUDIBLE: after the grudge TTS succeeds, the
# SAME persona voice speaks lines.mutter inside the SAME GPU window, when
# enough conservative margin remains. Worst-case math (the latency truth):
# the mutter is only attempted while elapsed <= 63.0 s (75 - 12). A
# WRITING.md mutter is <=10 words ~= 2-4 s of audio; at VoxCPM2's verified
# RTF 0.5-1.0 (§7) that is ~2-4 s of GPU, so the worst case completes by
# ~67 s — under ~70 s even with a couple seconds of variance, with the 75 s
# window never threatened. p50 leaves ~50 s of margin (~25 s used of 75), so
# in practice the mutter almost always speaks.
MUTTER_MIN_MARGIN_SECONDS = 12.0
def mutter_window_open(elapsed_seconds: float) -> bool:
"""True while >= MUTTER_MIN_MARGIN_SECONDS of the GPU window remain."""
return (GPU_WINDOW_SECONDS - elapsed_seconds) >= MUTTER_MIN_MARGIN_SECONDS
# --- Geometry-retry window guard -----------------------------------------------
# The geometry retry (June 12 audit: cramped eye pairs, busy-texture drift,
# edge picks) is a full second VLM generation — roughly the séance's own cost
# — so its guard is the most conservative of the three: it must leave room
# for itself AND the TTS that follows. Under margin the retry is skipped
# (never the awakening) and the initial features ship with violations logged.
GEOMETRY_MIN_MARGIN_SECONDS = 25.0
def geometry_window_open(elapsed_seconds: float) -> bool:
"""True while >= GEOMETRY_MIN_MARGIN_SECONDS of the GPU window remain."""
return (GPU_WINDOW_SECONDS - elapsed_seconds) >= GEOMETRY_MIN_MARGIN_SECONDS
# Full-line code fences only — same shape as mind/schema.py uses for JSON.
_PUNCHUP_FENCE_LINE = re.compile(r"^\s*```[a-zA-Z0-9_-]*\s*$", re.MULTILINE)
#: Quote characters a chatty model may wrap the bare line in.
_PUNCHUP_QUOTE_CHARS = "\"'“”‘’`"
def strip_punchup_reply(raw: str) -> str:
"""Reduce a punch-up reply to the bare line (pure string work, no ML).
Drops code-fence lines, takes the first non-empty line (the model was told
"ONLY the line" — anything after a newline is commentary), strips wrapping
quotes/backticks, collapses whitespace. Returns "" when nothing survives;
the acceptance checker rejects "" and the original grudge ships.
"""
text = _PUNCHUP_FENCE_LINE.sub("", str(raw or ""))
for line in text.splitlines():
cleaned = " ".join(line.strip().strip(_PUNCHUP_QUOTE_CHARS).strip().split())
if cleaned:
return cleaned
return ""
class ZeroGPUMedium:
"""MiniCPM-V-4_5 spirit medium for the ZeroGPU Space (§7 pins).
Heavy rules, all verified June 12:
- attn_implementation='sdpa' (NEVER 'eager'; no flash-attn on the image),
torch_dtype=bfloat16, trust_remote_code=True, .eval().cuda().
- Load happens ONCE at construction. app.py must construct this medium at
module import (startup), where the ZeroGPU runtime manages the startup
.cuda() — that is the moral equivalent of §7's "module level" load, kept
inside this class so mock mode never imports torch.
- No native JSON mode: sampling=False + defensive parse + ONE repair retry
(schema.build_repair_prompt), then PoeticError. max_slice_nums=4 bounds
prefill at our ≤1024px inputs.
``chat_fn`` is a test seam: a callable ``(image, prompt) -> str`` that
replaces the loaded model so the retry/PoeticError logic is testable with
zero ML imports. ``chat_msgs_fn`` is the matching seam for the punch-up's
multi-turn call: ``(msgs, max_new_tokens) -> str``. When ``chat_fn`` is
given without ``chat_msgs_fn`` the punch-up simply never applies.
"""
name = "zerogpu"
DEFAULT_MODEL = "openbmb/MiniCPM-V-4_5"
MAX_NEW_TOKENS = 700
MAX_SLICE_NUMS = 4
# The punch-up rewrite is one short line: 60 tokens is ~3x the 22-word cap.
PUNCHUP_MAX_NEW_TOKENS = 60
def __init__(
self,
model_id: Optional[str] = None,
chat_fn: Optional[Callable[[Any, str], str]] = None,
chat_msgs_fn: Optional[Callable[[list, int], str]] = None,
):
self.model_id = model_id or os.environ.get("PAREIDOLIA_VLM", self.DEFAULT_MODEL)
if chat_fn is not None:
self._chat: Callable[[Any, str], str] = chat_fn
self._chat_msgs: Optional[Callable[[list, int], str]] = chat_msgs_fn
else:
self._chat, self._chat_msgs = self._load()
# (prompt, raw_reply) of the most recent ACCEPTED parse — the punch-up
# and the geometry retry replay it as conversation history. One
# awakening per medium at a time by construction (the ZeroGPU worker
# serializes GPU calls).
self._last_exchange: Optional[tuple[str, str]] = None
# True once this awakening's single repair retry went to a parse
# failure — the geometry retry then stands down (one extra generation
# per awakening, total). Reset at the top of awaken().
self._repair_retry_burned = False
def _load(self) -> tuple[Callable[[Any, str], str], Callable[[list, int], str]]:
"""Load MiniCPM-V once; return (chat, chat_msgs) bound callables.
The ONLY torch/transformers imports in the mind package live here.
"""
import torch # noqa: PLC0415 - guarded heavy import by design
from transformers import AutoModel, AutoTokenizer # noqa: PLC0415
tokenizer = AutoTokenizer.from_pretrained(self.model_id, trust_remote_code=True)
model = AutoModel.from_pretrained(
self.model_id,
trust_remote_code=True,
attn_implementation="sdpa",
torch_dtype=torch.bfloat16,
)
model = model.eval().cuda()
def _chat(image: Any, prompt: str) -> str:
return model.chat(
msgs=[{"role": "user", "content": [image, prompt]}],
tokenizer=tokenizer,
sampling=False,
max_new_tokens=self.MAX_NEW_TOKENS,
max_slice_nums=self.MAX_SLICE_NUMS,
# MiniCPM-V-4_5 is a hybrid-thinking model: thinking MUST stay
# off (§7, bench-proven call) or thinking tokens eat the
# 700-token budget and the JSON truncates every time.
enable_thinking=False,
)
def _chat_msgs(msgs: list, max_new_tokens: int) -> str:
# The punch-up call: full multi-turn history, sampling ON (greedy
# is exactly what made the grudges template-y), tiny budget.
return model.chat(
msgs=msgs,
tokenizer=tokenizer,
sampling=True,
max_new_tokens=max_new_tokens,
max_slice_nums=self.MAX_SLICE_NUMS,
enable_thinking=False,
)
return _chat, _chat_msgs
def awaken(self, image: Any, prompt: Optional[str] = None) -> AwakeningResult:
"""One structured generation + at most one repair retry.
Raises PoeticError (visitor-safe copy) when both attempts fail; the
underlying parse errors ride on __cause__ for the logs. The accepted
(prompt, raw reply) pair is kept on the instance for punch_up().
"""
seance = prompt or SEANCE_PROMPT
self._last_exchange = None
self._repair_retry_burned = False
raw = self._chat(image, seance)
try:
result = parse_awakening(raw)
self._last_exchange = (seance, raw)
return result
except AwakeningParseError as first_error:
logger.warning("awakening parse failed, retrying once: %s", first_error)
self._repair_retry_burned = True # the one extra generation is spent
repair = build_repair_prompt(str(first_error), raw)
raw_retry = self._chat(image, repair)
try:
# retry=True is the soft-accept: an eye-separation-only failure
# on the SECOND parse is accepted with a logged warning
# (overlay.js's render-side spread clamp displays tight pairs
# safely) — a flat-but-servable record beats a PoeticError.
# June 12: this saves the coffee-mug class of double refusals.
result = parse_awakening(raw_retry, retry=True)
self._last_exchange = (repair, raw_retry)
return result
except AwakeningParseError as second_error:
logger.error("awakening repair also failed: %s", second_error)
raise PoeticError(REFUSAL_SCATTERED) from second_error
@property
def repair_retry_burned(self) -> bool:
"""True when this awakening already spent its repair retry on a parse
failure — _pipeline's geometry retry checks this so one awakening
never costs more than one extra generation, whatever the reason."""
return self._repair_retry_burned
def repair_geometry(
self, image: Any, violations: list[str]
) -> Optional[AwakeningResult]:
"""ONE multi-turn retry asking for different features for failing roles.
The proven repair mechanism aimed at face geometry: replay the
accepted exchange (image, séance prompt, the model's own raw reply),
then complain with the post-snap violations verbatim
(schema.build_geometry_repair_prompt) and parse the reply with
retry=True semantics. Rides the same sampled multi-turn callable as
the punch-up — sampling is exactly right here, since greedy decoding
over near-identical context tends to reproduce the same feature picks.
Returns the parsed candidate, or None when the reply does not parse
(or the multi-turn seam is unavailable) — the caller keeps the initial
feature set either way; geometry NEVER refuses an awakening. The
caller ignores the returned persona/lines: only features may be
replaced (writing quality is eval-tuned and the first parse won it).
"""
if self._chat_msgs is None or self._last_exchange is None:
return None
prompt_used, raw_reply = self._last_exchange
msgs = [
{"role": "user", "content": [image, prompt_used]},
{"role": "assistant", "content": [raw_reply]},
{"role": "user", "content": [build_geometry_repair_prompt(violations)]},
]
raw = self._chat_msgs(msgs, self.MAX_NEW_TOKENS)
try:
return parse_awakening(raw, retry=True)
except AwakeningParseError as exc:
logger.warning(
"geometry retry reply did not parse; keeping the initial "
"features: %s",
exc,
)
return None
def punch_up(self, image: Any, result: AwakeningResult) -> dict:
"""One short SAMPLED rewrite of the grudge line, strictly validated.
Runs inside the same GPU window, after the gate passed and BEFORE TTS
(the rewritten line is what gets spoken). The multi-turn msgs replay
the original exchange — image, séance prompt, the model's own raw
JSON reply — then ask for the rewrite (prompts.build_punchup_prompt).
ACCEPTS the rewrite only when prompts.punchup_rejection returns None
(LINE_VALIDATORS pass, 8-22 words, differs from the original, cites a
visible specific); then mutates ``result.lines.grudge`` in place.
Every other outcome keeps the original line. Returns the provenance
dict for ``record["punchup"]``: {"applied": bool, "original": str|None}.
"""
not_applied = {"applied": False, "original": None}
if (
self._chat_msgs is None
or self._last_exchange is None
or _build_punchup_prompt is None
or _punchup_rejection is None
):
return not_applied
original = result.lines.grudge
record = result.model_dump()
prompt_used, raw_reply = self._last_exchange
msgs = [
{"role": "user", "content": [image, prompt_used]},
{"role": "assistant", "content": [raw_reply]},
{"role": "user", "content": [_build_punchup_prompt(record)]},
]
raw = self._chat_msgs(msgs, self.PUNCHUP_MAX_NEW_TOKENS)
candidate = strip_punchup_reply(raw)
reason = _punchup_rejection(candidate, original, record)
if reason is not None:
logger.info("punch-up rejected (%s); keeping the original grudge", reason)
return not_applied
result.lines.grudge = candidate
return {"applied": True, "original": original}
# ---------------------------------------------------------------------------
# Factory
# ---------------------------------------------------------------------------
_MEDIUM_ALIASES = {
"mock": "mock",
"": "mock",
"zerogpu": "zerogpu",
"zero-gpu": "zerogpu",
"minicpm": "zerogpu",
}
def make_medium(backend_name: Optional[str] = None):
"""Build the medium for ``backend_name`` or the PAREIDOLIA_BACKEND env.
Defaults to mock (always works, zero deps beyond pydantic).
"""
raw = (backend_name or os.environ.get("PAREIDOLIA_BACKEND") or "mock").strip().lower()
resolved = _MEDIUM_ALIASES.get(raw)
if resolved == "mock":
return MockMedium()
if resolved == "zerogpu":
return ZeroGPUMedium()
raise ValueError(f"unknown PAREIDOLIA_BACKEND {raw!r} (expected mock | zerogpu)")
# ---------------------------------------------------------------------------
# The combined GPU window — VLM -> snap -> TTS in one quota spend (§2)
# ---------------------------------------------------------------------------
def _pipeline(
image: Any,
medium,
voice,
prompt: Optional[str] = None,
window_start: Optional[float] = None,
) -> dict:
"""The full awakening pipeline, backend-agnostic.
Returns the dict app.py shapes into the §2 response:
{
"refused": bool,
"refusal": str | None, # poetic copy when refused
"record": dict | None, # AwakeningResult dump + "features" (post-
# snap, each with snap_delta; the original
# candidate_features stay for the trace
# unless the geometry retry's set won) +
# "geometry" + "punchup" provenance (live
# medium only)
"grudge_wav": bytes | None, # WAV bytes; app.py b64-encodes
"mutter_wav": bytes | None, # same persona voice; None on skip or
# mutter-TTS failure — best-effort by
# contract (refusal dicts, which exit
# before all TTS, omit the key; use
# .get())
}
Refusals exit BEFORE snap, punch-up, and TTS — the gate is also the
budget guard.
``window_start`` is the GPU window's time.monotonic() origin (passed by
_gpu_entry); it feeds the punch-up latency guard. Defaults to "now" for
direct callers.
"""
if window_start is None:
window_start = time.monotonic()
result: AwakeningResult = (
medium.awaken(image, prompt=prompt) if prompt is not None else medium.awaken(image)
)
refusal = result.gate.refusal()
if refusal is not None:
return {"refused": True, "refusal": refusal, "record": None, "grudge_wav": None}
coarse = [feature.model_dump() for feature in result.candidate_features]
try:
snapped = _snap_features(_as_snap_image(image), coarse)
except Exception: # noqa: BLE001 - snapping is best-effort by contract (§0)
logger.exception("cv snap failed; keeping the VLM's coarse points")
snapped = [{**feature, "snap_delta": 0.0} for feature in coarse]
# Geometry gate (June 12 audit: cramped eye pairs, busy-texture drift,
# edge picks): score the FINAL post-snap geometry; on violations, ask the
# model ONCE for different features for the failing roles — multi-turn,
# image still in context — unless this awakening already burned its
# repair retry on a parse failure or the window margin is too thin.
# Whichever candidate set scores FEWER violations ships (tie -> the
# retry: it was asked to fix things); persona/lines from the FIRST
# accepted parse always win — only features may be replaced. Geometry
# NEVER refuses an awakening: worst case the better-scoring set ships
# with its violations logged. Only media exposing repair_geometry (the
# live ZeroGPUMedium) participate — mock records stay byte-identical.
geometry: Optional[dict] = None
winning_coarse: Optional[list] = None
repair_geometry = getattr(medium, "repair_geometry", None)
if callable(repair_geometry):
violations = face_geometry_violations(snapped)
geometry = {
"violations_initial": violations,
"retried": False,
"violations_final": violations,
}
if violations and getattr(medium, "repair_retry_burned", False):
logger.info(
"geometry retry skipped: the repair retry is already spent on "
"a parse failure; keeping the initial features"
)
elif violations:
elapsed = time.monotonic() - window_start
if not geometry_window_open(elapsed):
logger.info(
"geometry retry skipped: %.1fs elapsed of the %.0fs window "
"leaves under %.0fs margin",
elapsed,
GPU_WINDOW_SECONDS,
GEOMETRY_MIN_MARGIN_SECONDS,
)
else:
geometry["retried"] = True
t0 = time.monotonic()
alt: Optional[AwakeningResult] = None
try:
alt = repair_geometry(image, violations)
except Exception: # noqa: BLE001 - never costs the awakening
logger.exception(
"geometry retry failed; keeping the initial features"
)
if alt is not None:
alt_coarse = [f.model_dump() for f in alt.candidate_features]
try:
alt_snapped = _snap_features(_as_snap_image(image), alt_coarse)
except Exception: # noqa: BLE001 - best-effort (§0)
logger.exception(
"cv snap failed on the geometry retry; keeping its "
"coarse points"
)
alt_snapped = [
{**feature, "snap_delta": 0.0} for feature in alt_coarse
]
alt_violations = face_geometry_violations(alt_snapped)
if len(alt_violations) <= len(violations):
snapped = alt_snapped
winning_coarse = alt_coarse
geometry["violations_final"] = alt_violations
logger.info(
"geometry retry %s in %.1fs (%d -> %d violation(s))",
"replaced the features" if winning_coarse is not None
else "kept the initial features",
time.monotonic() - t0,
len(violations),
len(geometry["violations_final"]),
)
if geometry["violations_final"]:
logger.warning(
"shipping with %d geometry violation(s): %s",
len(geometry["violations_final"]),
"; ".join(geometry["violations_final"]),
)
# Punch-up: second short sampled generation that rewrites ONLY the grudge
# line, BEFORE TTS so the rewritten line is what gets spoken. Only media
# exposing punch_up (the live ZeroGPUMedium) participate — the mock has no
# model and its records stay byte-identical to before.
# OFF BY DEFAULT (PAREIDOLIA_PUNCHUP=1 to enable): the June-12 live A/B
# (6 images, prod) showed the acceptance bar admits grammatically broken
# rewrites — 0/4 accepted rewrites beat the original. The bar needs a
# coherence gate (finite verb, no contradiction with the stated condition)
# before this flag flips.
punchup: Optional[dict] = None
punch = getattr(medium, "punch_up", None)
if os.environ.get("PAREIDOLIA_PUNCHUP", "0") != "1":
punch = None
if callable(punch):
elapsed = time.monotonic() - window_start
if not punchup_window_open(elapsed):
logger.info(
"punch-up skipped: %.1fs elapsed of the %.0fs window leaves "
"under %.0fs margin",
elapsed,
GPU_WINDOW_SECONDS,
PUNCHUP_MIN_MARGIN_SECONDS,
)
punchup = {"applied": False, "original": None}
else:
t0 = time.monotonic()
try:
punchup = punch(image, result) # mutates result.lines.grudge on accept
except Exception: # noqa: BLE001 - polish must never cost the awakening
logger.exception("punch-up failed; keeping the original grudge")
punchup = {"applied": False, "original": None}
logger.info(
"punch-up %s in %.1fs",
"applied" if punchup.get("applied") else "not applied",
time.monotonic() - t0,
)
grudge_wav = voice.speak(result.lines.grudge, result.persona.voice)
# Mutter TTS — the wall's idle mutters become audible. Same persona voice,
# same GPU window, strictly best-effort: runs only AFTER the grudge TTS
# succeeded, only while MUTTER_MIN_MARGIN_SECONDS of conservative margin
# remain (see the worst-case math at mutter_window_open), and any failure
# is logged and swallowed — the awakening NEVER fails because of the
# mutter.
mutter_wav: Optional[bytes] = None
if (result.lines.mutter or "").strip():
elapsed = time.monotonic() - window_start
if not mutter_window_open(elapsed):
logger.info(
"mutter TTS skipped: %.1fs elapsed of the %.0fs window leaves "
"under %.0fs margin",
elapsed,
GPU_WINDOW_SECONDS,
MUTTER_MIN_MARGIN_SECONDS,
)
else:
t0 = time.monotonic()
try:
mutter_wav = voice.speak(result.lines.mutter, result.persona.voice)
logger.info("mutter TTS done in %.1fs", time.monotonic() - t0)
except Exception: # noqa: BLE001 - the mutter never costs the awakening
logger.exception(
"mutter TTS failed after %.1fs; awakening proceeds without it",
time.monotonic() - t0,
)
mutter_wav = None
record = result.model_dump()
record["features"] = snapped
if winning_coarse is not None:
# The geometry retry's features won — keep the pre-snap trace coherent
# with what actually renders. Everything else (object, persona, lines,
# critique) stays from the FIRST accepted parse.
record["candidate_features"] = winning_coarse
if geometry is not None:
record["geometry"] = geometry
if punchup is not None:
record["punchup"] = punchup
return {
"refused": False,
"refusal": None,
"record": record,
"grudge_wav": grudge_wav,
"mutter_wav": mutter_wav,
}
# ZeroGPU pickles every ARGUMENT of a @spaces.GPU call into the forked worker
# (spaces/utils.py arg_queue) — medium/voice hold closures over the loaded
# models and are unpicklable (PicklingError: "Can't pickle local object
# 'ZeroGPUMedium._load.<locals>._chat'", live-verified June 12). So the GPU
# entry takes ONLY picklable args; the actors are resolved from module state,
# which the worker inherits via fork. register_gpu_actors() must run at app
# startup (MindPipeline.__init__), before the first GPU call forks the worker.
_GPU_ACTORS: dict = {"medium": None, "voice": None}
def register_gpu_actors(medium, voice) -> None:
"""Pin the process-wide actors the GPU worker will use (pre-fork)."""
_GPU_ACTORS["medium"] = medium
_GPU_ACTORS["voice"] = voice
# Decorate at import time: ZeroGPU discovers @spaces.GPU functions at app
# startup, and importing `spaces` here keeps it ahead of torch (§7).
# Locally (no `spaces`) the bare entry stands in.
def _gpu_entry(image: Any, prompt: Optional[str] = None) -> dict:
window_start = time.monotonic() # origin for the punch-up latency guard
medium = _GPU_ACTORS["medium"]
voice = _GPU_ACTORS["voice"]
if medium is None or voice is None: # last-resort: loads INSIDE the window
logger.warning("gpu actors unregistered; building defaults in-window")
medium, voice = _defaults()
return _pipeline(image, medium, voice, prompt, window_start=window_start)
try: # pragma: no cover - `spaces` exists only on the HF runtime
import spaces # noqa: PLC0415 - MUST import before torch (§7 hard rule)
_gpu_pipeline: Callable[..., dict] = spaces.GPU(duration=75)(_gpu_entry)
except Exception: # noqa: BLE001 - local dev / tests
_gpu_pipeline = _gpu_entry
_DEFAULTS_LOCK = threading.Lock()
_DEFAULT_MEDIUM = None
_DEFAULT_VOICE = None
def _defaults():
"""Process-wide medium/voice singletons resolved from PAREIDOLIA_BACKEND.
Cached so the zerogpu path loads its models exactly once — app.py may
simply call awaken_full(image) per request. Construct eagerly at startup
(e.g. `mind.backends.warm_defaults()` from app.py's module level) when
running on ZeroGPU so weights are resident before the first visitor.
"""
global _DEFAULT_MEDIUM, _DEFAULT_VOICE
with _DEFAULTS_LOCK:
if _DEFAULT_MEDIUM is None:
_DEFAULT_MEDIUM = make_medium()
if _DEFAULT_VOICE is None:
_DEFAULT_VOICE = make_voice()
return _DEFAULT_MEDIUM, _DEFAULT_VOICE
def warm_defaults() -> None:
"""Eagerly build the default medium + voice (call at app startup)."""
medium, voice = _defaults()
preload = getattr(voice, "preload", None)
if callable(preload):
preload()
del medium
def awaken_full(
image: Any,
*,
medium=None,
voice=None,
prompt: Optional[str] = None,
) -> dict:
"""Awaken one object end to end: VLM -> gate -> CV snap -> TTS.
This is the function app.py's `awaken` API calls. On the zerogpu backend
the whole pipeline runs inside ONE @spaces.GPU(duration=75) window — one
queue wait, one quota spend, charged to the visitor's own browser-
authenticated gradio request (§0). Mock (and any injected test medium)
runs the bare pipeline: no spaces, no torch, no GPU.
Raises PoeticError when the spirits decline twice; every other outcome —
including gate refusals — is a normal dict (see _pipeline).
"""
if medium is None or voice is None:
default_medium, default_voice = _defaults()
medium = medium or default_medium
voice = voice or default_voice
if getattr(medium, "name", "") == "zerogpu":
# Only image+prompt may cross the pickle boundary into the GPU worker;
# the actors ride module state (see register_gpu_actors above).
register_gpu_actors(medium, voice)
return _gpu_pipeline(image, prompt)
return _pipeline(image, medium, voice, prompt)