"""Pure helpers for the on-pod scene-pipeline validation report. Kept dependency-free (operates on ``VideoResult.to_dict()`` dicts) so the report logic can be unit-tested on a laptop without loading any GPU models. The GPU orchestration lives in ``scripts/validate-scene-pipeline.py``. """ from __future__ import annotations from typing import Any def summarize_video_result(result: dict[str, Any]) -> dict[str, Any]: """Distill one video's pipeline output into validation metrics.""" windows = result.get("windows", []) or [] n = len(windows) clap_windows = sum(1 for w in windows if w.get("audio_tags")) onscreen = sum(len(w.get("onscreen_text_places") or []) for w in windows) window_loc_hits = sum(len(w.get("window_locations") or []) for w in windows) summary = result.get("video_summary") or {} locations = result.get("video_locations") or {} places = locations.get("places") or [] isos = sorted({p.get("country_iso") for p in places if p.get("country_iso")}) return { "natural_key": result.get("natural_key", ""), "windows": n, "clap_windows": clap_windows, "clap_coverage": round(clap_windows / n, 3) if n else 0.0, "onscreen_text_count": onscreen, "summary_parse_status": summary.get("parse_status"), "tldr_words": len((summary.get("tldr") or "").split()), "themes": len(summary.get("themes") or []), "acts": len(summary.get("acts") or []), "locations": len(places), "location_isos": isos, "window_location_hits": window_loc_hits, } def verdict(metrics: list[dict[str, Any]]) -> tuple[bool, list[str]]: """Smoke-test PASS/FAIL across all videos. Returns (passed, reasons). The headline gate is CLAP: the whole point of the sigmoid fix was to turn the old 0/175 coverage non-zero. Summaries parsing and at least one valid location are strong-but-soft signals (a clip may genuinely name no place). """ reasons: list[str] = [] if not metrics: return False, ["no videos were processed"] clap_total = sum(m["clap_windows"] for m in metrics) summaries_ok = sum(1 for m in metrics if m["summary_parse_status"] == "ok") loc_videos = sum(1 for m in metrics if m["locations"] > 0) any_valid_iso = any(m["location_isos"] for m in metrics) passed = True if clap_total > 0: reasons.append(f"PASS CLAP: {clap_total} tagged windows (was 0 before the fix)") else: passed = False reasons.append("FAIL CLAP: 0 tagged windows across all videos") if summaries_ok > 0: reasons.append(f"PASS summaries: {summaries_ok}/{len(metrics)} parsed cleanly") else: passed = False reasons.append("FAIL summaries: none parsed as clean JSON") if loc_videos > 0 and any_valid_iso: reasons.append(f"PASS locations: {loc_videos} video(s) with valid ISO codes") else: reasons.append("INFO locations: none found — OK only if these clips name no places") return passed, reasons def format_report(metrics: list[dict[str, Any]], sweep: list[dict[str, Any]] | None = None) -> str: """Render a compact text report for the pod terminal.""" lines = ["", "=" * 64, "SCENE-PIPELINE VALIDATION", "=" * 64] for m in metrics: lines.append(f"\nā–¶ {m['natural_key']} ({m['windows']} windows)") lines.append( f" CLAP coverage : {m['clap_windows']}/{m['windows']} " f"({m['clap_coverage']*100:.0f}%)" ) lines.append( f" summary : status={m['summary_parse_status']} " f"tldr={m['tldr_words']}w themes={m['themes']} acts={m['acts']}" ) lines.append( f" locations : {m['locations']} places " f"iso={m['location_isos']} window-hits={m['window_location_hits']}" ) lines.append(f" on-screen text: {m['onscreen_text_count']} mentions") if sweep: lines.append("\n" + "-" * 64) lines.append("THROUGHPUT SWEEP (--vlm-batch-size)") lines.append(f" {'batch':>6} {'win/s':>8} {'peak VRAM GB':>13}") for s in sweep: vram = f"{s['peak_vram_gb']:.1f}" if s.get("peak_vram_gb") is not None else "n/a" lines.append(f" {s['batch_size']:>6} {s['windows_per_sec']:>8.2f} {vram:>13}") passed, reasons = verdict(metrics) lines.append("\n" + "-" * 64) for r in reasons: lines.append(" " + r) lines.append("-" * 64) lines.append(f"OVERALL: {'PASS āœ…' if passed else 'FAIL āŒ'}") lines.append("=" * 64) return "\n".join(lines)