jw-search / backend /scene_processing /validation_report.py
jw-tools's picture
deploy: latest main (lazy-ML cold start, durable launcher, web-image search, scene search) + full-app data refresh
7ea1851 verified
Raw
History Blame Contribute Delete
4.65 kB
"""Pure helpers for the on-pod scene-pipeline validation report.
Kept dependency-free (operates on ``VideoResult.to_dict()`` dicts) so the
report logic can be unit-tested on a laptop without loading any GPU models.
The GPU orchestration lives in ``scripts/validate-scene-pipeline.py``.
"""
from __future__ import annotations
from typing import Any
def summarize_video_result(result: dict[str, Any]) -> dict[str, Any]:
"""Distill one video's pipeline output into validation metrics."""
windows = result.get("windows", []) or []
n = len(windows)
clap_windows = sum(1 for w in windows if w.get("audio_tags"))
onscreen = sum(len(w.get("onscreen_text_places") or []) for w in windows)
window_loc_hits = sum(len(w.get("window_locations") or []) for w in windows)
summary = result.get("video_summary") or {}
locations = result.get("video_locations") or {}
places = locations.get("places") or []
isos = sorted({p.get("country_iso") for p in places if p.get("country_iso")})
return {
"natural_key": result.get("natural_key", ""),
"windows": n,
"clap_windows": clap_windows,
"clap_coverage": round(clap_windows / n, 3) if n else 0.0,
"onscreen_text_count": onscreen,
"summary_parse_status": summary.get("parse_status"),
"tldr_words": len((summary.get("tldr") or "").split()),
"themes": len(summary.get("themes") or []),
"acts": len(summary.get("acts") or []),
"locations": len(places),
"location_isos": isos,
"window_location_hits": window_loc_hits,
}
def verdict(metrics: list[dict[str, Any]]) -> tuple[bool, list[str]]:
"""Smoke-test PASS/FAIL across all videos. Returns (passed, reasons).
The headline gate is CLAP: the whole point of the sigmoid fix was to turn
the old 0/175 coverage non-zero. Summaries parsing and at least one valid
location are strong-but-soft signals (a clip may genuinely name no place).
"""
reasons: list[str] = []
if not metrics:
return False, ["no videos were processed"]
clap_total = sum(m["clap_windows"] for m in metrics)
summaries_ok = sum(1 for m in metrics if m["summary_parse_status"] == "ok")
loc_videos = sum(1 for m in metrics if m["locations"] > 0)
any_valid_iso = any(m["location_isos"] for m in metrics)
passed = True
if clap_total > 0:
reasons.append(f"PASS CLAP: {clap_total} tagged windows (was 0 before the fix)")
else:
passed = False
reasons.append("FAIL CLAP: 0 tagged windows across all videos")
if summaries_ok > 0:
reasons.append(f"PASS summaries: {summaries_ok}/{len(metrics)} parsed cleanly")
else:
passed = False
reasons.append("FAIL summaries: none parsed as clean JSON")
if loc_videos > 0 and any_valid_iso:
reasons.append(f"PASS locations: {loc_videos} video(s) with valid ISO codes")
else:
reasons.append("INFO locations: none found — OK only if these clips name no places")
return passed, reasons
def format_report(metrics: list[dict[str, Any]], sweep: list[dict[str, Any]] | None = None) -> str:
"""Render a compact text report for the pod terminal."""
lines = ["", "=" * 64, "SCENE-PIPELINE VALIDATION", "=" * 64]
for m in metrics:
lines.append(f"\n▶ {m['natural_key']} ({m['windows']} windows)")
lines.append(
f" CLAP coverage : {m['clap_windows']}/{m['windows']} "
f"({m['clap_coverage']*100:.0f}%)"
)
lines.append(
f" summary : status={m['summary_parse_status']} "
f"tldr={m['tldr_words']}w themes={m['themes']} acts={m['acts']}"
)
lines.append(
f" locations : {m['locations']} places "
f"iso={m['location_isos']} window-hits={m['window_location_hits']}"
)
lines.append(f" on-screen text: {m['onscreen_text_count']} mentions")
if sweep:
lines.append("\n" + "-" * 64)
lines.append("THROUGHPUT SWEEP (--vlm-batch-size)")
lines.append(f" {'batch':>6} {'win/s':>8} {'peak VRAM GB':>13}")
for s in sweep:
vram = f"{s['peak_vram_gb']:.1f}" if s.get("peak_vram_gb") is not None else "n/a"
lines.append(f" {s['batch_size']:>6} {s['windows_per_sec']:>8.2f} {vram:>13}")
passed, reasons = verdict(metrics)
lines.append("\n" + "-" * 64)
for r in reasons:
lines.append(" " + r)
lines.append("-" * 64)
lines.append(f"OVERALL: {'PASS ✅' if passed else 'FAIL ❌'}")
lines.append("=" * 64)
return "\n".join(lines)