"""Visual analysis of sampled frames. Primary method: **YOLO** (Ultralytics, e.g. ``yolo11n``) — a tiny object detector that runs comfortably on Apple-Silicon CPU. It yields COCO classes such as ``person``, ``laptop``, ``tv`` (screen-like), ``cell phone``, ``book``, which are exactly the kinds of objects that appear in tutorial / lecture / presentation videos. Augmentation: a lightweight **scene-change** signal from frame-to-frame colour histogram correlation (OpenCV). Large drops flag slide changes / cuts — useful for bookmarking even when no object class is informative. Fallback: if Ultralytics is not installed, the module still produces a scene-change-only timeline so the rest of the pipeline keeps working. The chosen method is recorded in each result via the ``method`` field. All heavy imports are lazy. """ from __future__ import annotations from collections import Counter from typing import Dict, List, Optional from src.config import Config, CONFIG from src.utils import format_timestamp # COCO classes we treat as "screen / slide-like" surfaces. SCREEN_LIKE = {"tv", "laptop", "cell phone", "monitor", "book"} def _try_load_yolo(model_name: str): """Return a loaded YOLO model or ``None`` if Ultralytics is unavailable.""" try: from ultralytics import YOLO # type: ignore except ImportError: return None try: return YOLO(model_name) except Exception: return None def _histogram(cv2, image): hist = cv2.calcHist([image], [0, 1, 2], None, [8, 8, 8], [0, 256] * 3) cv2.normalize(hist, hist) return hist.flatten() def analyze_frames( frame_paths: List[str], frame_times: List[float], config: Config = CONFIG, ) -> List[Dict[str, object]]: """Detect objects + scene changes for each sampled frame. Parameters mirror the output of :func:`video_preprocessing.extract_frames`. Returns one record per frame: { "time_sec": 42.0, "time_label": "00:00:42", "visual_events": ["person", "laptop", "screen"], "confidence": {"person": 0.94, "laptop": 0.82}, "scene_change": true, "method": "yolo11n.pt+scene" } """ try: import cv2 # type: ignore except ImportError as exc: # pragma: no cover raise RuntimeError( "opencv-python is required for visual analysis. " "Install with `pip install -r requirements-local.txt`." ) from exc model = _try_load_yolo(config.visual_model) method = ( f"{config.visual_model}+scene" if model is not None else "scene-change-only" ) results: List[Dict[str, object]] = [] prev_hist = None for path, t in zip(frame_paths, frame_times): image = cv2.imread(path) if image is None: continue # ---- scene change via histogram correlation --------------------- # hist = _histogram(cv2, image) scene_change = False if prev_hist is not None: corr = float(cv2.compareHist(prev_hist, hist, cv2.HISTCMP_CORREL)) scene_change = (1.0 - corr) >= config.scene_change_threshold prev_hist = hist # ---- object detection ------------------------------------------- # events: List[str] = [] confidence: Dict[str, float] = {} if model is not None: preds = model.predict( image, conf=config.visual_conf_threshold, verbose=False ) for pred in preds: names = pred.names for box in pred.boxes: cls_id = int(box.cls[0]) label = names.get(cls_id, str(cls_id)) conf = float(box.conf[0]) # keep the highest confidence per label in this frame if conf > confidence.get(label, 0.0): confidence[label] = round(conf, 3) events = sorted(confidence, key=confidence.get, reverse=True) # Add a derived "screen" concept when a screen-like object is seen. if any(e in SCREEN_LIKE for e in events) and "screen" not in events: events.append("screen") # NOTE: scene changes are reported via the dedicated `scene_change` # boolean below (not mixed into the object list), so downstream code # has a single source of truth. results.append( { "time_sec": round(float(t), 3), "time_label": format_timestamp(t), "visual_events": events, "confidence": confidence, "scene_change": scene_change, "method": method, } ) return results def object_frequency( visual_events: List[Dict[str, object]], top_k: Optional[int] = None, ) -> List[Dict[str, object]]: """Aggregate how often each visual concept appears across all frames.""" counter: Counter = Counter() for rec in visual_events: for ev in rec.get("visual_events", []): if ev == "scene_change": continue counter[ev] += 1 items = counter.most_common(top_k) return [{"label": label, "count": count} for label, count in items]