ai-video-intelligence-agent / src /video_source.py
pmootr's picture
Add URL/YouTube ingestion, "add-a-sample" workflow, and architecture doc
aa5e3cc
Raw
History Blame Contribute Delete
5.1 kB
"""Resolve a *video source* (local path **or** URL) to a local file to analyze.
Supports any site handled by `yt-dlp` (YouTube, Vimeo, direct links, ...), so
the pipeline can ingest an online learning video by URL instead of requiring a
local download first.
Design notes
------------
* `yt-dlp` is a **local-analysis** dependency (`requirements-local.txt`), imported
lazily — the lightweight cloud demo never needs it.
* Downloads are capped to **720p** to stay MacBook-Air-M4 friendly.
* The `MAX_VIDEO_DURATION_SEC` guard is enforced: a longer video is **clipped to
the first N seconds** (with a warning) rather than silently processing a huge
file — consistent with the "short videos only" assumption.
⚠️ **Usage / licensing:** only download videos you have the right to use
(Creative Commons, public domain, your own, or with permission), and respect each
platform's Terms of Service. This tool does not grant any rights to third-party
content.
"""
from __future__ import annotations
import re
import warnings
from pathlib import Path
from typing import Dict, Tuple
from src.config import Config, CONFIG
from src.storage import work_dir
from src.utils import slugify
_URL_RE = re.compile(r"^https?://", re.IGNORECASE)
_MEDIA_EXTS = {".mp4", ".mkv", ".webm", ".mov", ".m4v", ".avi"}
def is_url(source: str) -> bool:
"""True if ``source`` looks like an http(s) URL (vs a local path)."""
return bool(_URL_RE.match(str(source).strip()))
def _require_ytdlp():
try:
import yt_dlp # type: ignore
return yt_dlp
except ImportError as exc: # pragma: no cover - environment dependent
raise RuntimeError(
"yt-dlp is required to ingest a video from a URL. "
"Install it with `pip install -r requirements-local.txt`."
) from exc
def probe_url(url: str, config: Config = CONFIG) -> Dict[str, object]:
"""Fetch metadata for a URL without downloading the media."""
yt = _require_ytdlp()
opts = {"quiet": True, "skip_download": True, "noplaylist": True, "no_warnings": True}
with yt.YoutubeDL(opts) as ydl:
info = ydl.extract_info(url, download=False)
return {
"id": info.get("id"),
"title": info.get("title"),
"duration_sec": info.get("duration"),
"uploader": info.get("uploader"),
"webpage_url": info.get("webpage_url", url),
"license": info.get("license"),
}
def _download(url: str, video_id: str, config: Config, clip_to_max: bool = True) -> Path:
yt = _require_ytdlp()
out_dir = work_dir(video_id)
out_dir.mkdir(parents=True, exist_ok=True)
opts: Dict[str, object] = {
"quiet": True,
"no_warnings": True,
"noprogress": True,
"noplaylist": True,
"restrictfilenames": True,
"outtmpl": str(out_dir / "source.%(ext)s"),
# Keep it light: prefer <=720p, merge to mp4 when needed.
"format": "bestvideo[height<=720]+bestaudio/best[height<=720]/best",
"merge_output_format": "mp4",
}
# Duration guard -> clip to the first MAX_VIDEO_DURATION_SEC seconds.
info = probe_url(url, config)
duration = info.get("duration_sec") or 0
max_dur = config.max_video_duration_sec
if max_dur and duration and duration > max_dur:
if clip_to_max:
warnings.warn(
f"Video is {int(duration)}s (> MAX_VIDEO_DURATION_SEC={max_dur}); "
f"downloading only the first {max_dur}s."
)
opts["download_ranges"] = yt.utils.download_range_func(None, [(0, max_dur)])
opts["force_keyframes_at_cuts"] = True
else:
raise ValueError(
f"Video is {int(duration)}s, which exceeds "
f"MAX_VIDEO_DURATION_SEC={max_dur}."
)
with yt.YoutubeDL(opts) as ydl:
ydl.download([url])
candidates = [
p for p in out_dir.glob("source.*") if p.suffix.lower() in _MEDIA_EXTS
]
if not candidates:
raise RuntimeError(f"Download produced no media file for {url}")
# Largest file is the merged/best media.
return max(candidates, key=lambda p: p.stat().st_size)
def resolve_source(
source: str,
video_id: str = None,
config: Config = CONFIG,
) -> Tuple[Path, Dict[str, object]]:
"""Return ``(local_video_path, source_info)`` for a path or URL.
``source_info`` always carries a suggested ``video_id`` plus provenance the
pipeline records into the metrics artifact.
"""
if is_url(source):
meta = probe_url(source, config)
vid = video_id or slugify(str(meta.get("id") or meta.get("title") or "video"))
path = _download(source, vid, config)
meta.update({"video_id": vid, "source": "url", "source_url": source})
return path, meta
path = Path(source)
if not path.exists():
raise FileNotFoundError(f"Video not found: {path}")
return path, {
"video_id": video_id or slugify(path.stem),
"title": path.stem,
"source": "local",
"source_path": str(path),
}