"""Structured visual attributes for scene windows: shot type, setting, era. These let footage searches match on *how* a scene is filmed and *when/where* it is set — "aerial shots of a Kingdom Hall exterior", "modern-day disaster relief", "Bible-times scenes set indoors" — which a free-text description alone captures inconsistently. Single source of truth for the controlled vocabularies, normalization of the model's free text into those vocabularies, and the search phrase that gets folded into a window's embedded text so the existing semantic search picks the attributes up (no query-side changes needed). Both the vision pipeline (going forward) and the text-backfill pass import from here so the two can't drift. """ from __future__ import annotations # canonical value -> synonym-rich phrase woven into the embedded text. The # phrase is what makes the attribute searchable: a query like "aerial drone # shot" embeds near a window whose text contains "aerial drone overhead". SHOT_TYPES: dict[str, str] = { "aerial": "aerial drone overhead bird's-eye shot", "establishing": "wide establishing landscape shot", "medium": "medium shot", "close-up": "close-up tight shot", "pov": "point-of-view first-person shot", "talking-head": "interview talking-head piece-to-camera shot", "unclear": "", } SETTINGS: dict[str, str] = { "interior": "indoor interior", "exterior": "outdoor exterior", "mixed": "both indoor and outdoor", "unclear": "", } ERAS: dict[str, str] = { "ancient": "Bible times ancient biblical era", "medieval": "medieval middle ages", "early-1900s": "early 1900s early twentieth century historical", "mid-1900s": "mid 1900s mid twentieth century historical", "late-1900s": "1970s 1980s 1990s late twentieth century", "modern-day": "modern-day present-day contemporary", "unclear": "", } # Free-text variants the model may emit -> canonical key. Checked as substrings # (longest-first) so "close up" / "closeup" / "tight" all map to "close-up". _SHOT_ALIASES: tuple[tuple[str, str], ...] = ( ("drone", "aerial"), ("aerial", "aerial"), ("overhead", "aerial"), ("bird", "aerial"), ("tilt-shift", "aerial"), ("establishing", "establishing"), ("wide", "establishing"), ("landscape", "establishing"), ("vista", "establishing"), ("close-up", "close-up"), ("close up", "close-up"), ("closeup", "close-up"), ("tight", "close-up"), ("portrait", "close-up"), ("macro", "close-up"), ("point-of-view", "pov"), ("point of view", "pov"), ("pov", "pov"), ("first-person", "pov"), ("talking head", "talking-head"), ("talking-head", "talking-head"), ("piece to camera", "talking-head"), ("to camera", "talking-head"), ("interview", "talking-head"), ("presenter", "talking-head"), ("medium", "medium"), ) _SETTING_ALIASES: tuple[tuple[str, str], ...] = ( ("interior", "interior"), ("indoor", "interior"), ("inside", "interior"), ("exterior", "exterior"), ("outdoor", "exterior"), ("outside", "exterior"), ("mixed", "mixed"), ("both", "mixed"), ) _ERA_ALIASES: tuple[tuple[str, str], ...] = ( ("biblical", "ancient"), ("bible times", "ancient"), ("ancient", "ancient"), ("first century", "ancient"), ("antiquity", "ancient"), ("medieval", "medieval"), ("middle ages", "medieval"), ("early 1900", "early-1900s"), ("early-1900", "early-1900s"), ("early twentieth", "early-1900s"), ("1910", "early-1900s"), ("1920", "early-1900s"), ("1930", "early-1900s"), ("mid 1900", "mid-1900s"), ("mid-1900", "mid-1900s"), ("1940", "mid-1900s"), ("1950", "mid-1900s"), ("1960", "mid-1900s"), ("1970", "late-1900s"), ("1980", "late-1900s"), ("1990", "late-1900s"), ("late 1900", "late-1900s"), ("late-1900", "late-1900s"), ("late twentieth", "late-1900s"), ("modern", "modern-day"), ("present", "modern-day"), ("contemporary", "modern-day"), ("today", "modern-day"), ("2000", "modern-day"), ("2010", "modern-day"), ("2020", "modern-day"), ) def _normalize(raw: object, aliases: tuple[tuple[str, str], ...], valid: dict[str, str]) -> str: """Map the model's free text to a canonical key; 'unclear' when no match.""" text = str(raw or "").strip().lower() if not text: return "unclear" if text in valid: # already canonical return text for needle, canonical in aliases: if needle in text: return canonical return "unclear" def normalize_shot_type(raw: object) -> str: return _normalize(raw, _SHOT_ALIASES, SHOT_TYPES) def normalize_setting(raw: object) -> str: return _normalize(raw, _SETTING_ALIASES, SETTINGS) def normalize_era(raw: object) -> str: return _normalize(raw, _ERA_ALIASES, ERAS) def attribute_search_phrase(shot_type: str, setting: str, era: str) -> str: """Build the phrase appended to a window's description before embedding. Empty parts ('unclear') are omitted. Returns '' when nothing is known, so a window with no usable attributes embeds exactly as its description did. """ parts: list[str] = [] shot = SHOT_TYPES.get(shot_type or "unclear", "") setting_phrase = SETTINGS.get(setting or "unclear", "") era_phrase = ERAS.get(era or "unclear", "") if shot: parts.append(f"Camera: {shot}.") if setting_phrase: parts.append(f"Setting: {setting_phrase}.") if era_phrase: parts.append(f"Era: {era_phrase}.") return " ".join(parts) def embed_text(description: str, shot_type: str, setting: str, era: str) -> str: """The full text to embed for a window: description + attribute phrase.""" phrase = attribute_search_phrase(shot_type, setting, era) description = (description or "").strip() return f"{description} {phrase}".strip() if phrase else description