Spaces:
Sleeping
Sleeping
| """Resolve a *video source* (local path **or** URL) to a local file to analyze. | |
| Supports any site handled by `yt-dlp` (YouTube, Vimeo, direct links, ...), so | |
| the pipeline can ingest an online learning video by URL instead of requiring a | |
| local download first. | |
| Design notes | |
| ------------ | |
| * `yt-dlp` is a **local-analysis** dependency (`requirements-local.txt`), imported | |
| lazily — the lightweight cloud demo never needs it. | |
| * Downloads are capped to **720p** to stay MacBook-Air-M4 friendly. | |
| * The `MAX_VIDEO_DURATION_SEC` guard is enforced: a longer video is **clipped to | |
| the first N seconds** (with a warning) rather than silently processing a huge | |
| file — consistent with the "short videos only" assumption. | |
| ⚠️ **Usage / licensing:** only download videos you have the right to use | |
| (Creative Commons, public domain, your own, or with permission), and respect each | |
| platform's Terms of Service. This tool does not grant any rights to third-party | |
| content. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| import warnings | |
| from pathlib import Path | |
| from typing import Dict, Tuple | |
| from src.config import Config, CONFIG | |
| from src.storage import work_dir | |
| from src.utils import slugify | |
| _URL_RE = re.compile(r"^https?://", re.IGNORECASE) | |
| _MEDIA_EXTS = {".mp4", ".mkv", ".webm", ".mov", ".m4v", ".avi"} | |
| def is_url(source: str) -> bool: | |
| """True if ``source`` looks like an http(s) URL (vs a local path).""" | |
| return bool(_URL_RE.match(str(source).strip())) | |
| def _require_ytdlp(): | |
| try: | |
| import yt_dlp # type: ignore | |
| return yt_dlp | |
| except ImportError as exc: # pragma: no cover - environment dependent | |
| raise RuntimeError( | |
| "yt-dlp is required to ingest a video from a URL. " | |
| "Install it with `pip install -r requirements-local.txt`." | |
| ) from exc | |
| def probe_url(url: str, config: Config = CONFIG) -> Dict[str, object]: | |
| """Fetch metadata for a URL without downloading the media.""" | |
| yt = _require_ytdlp() | |
| opts = {"quiet": True, "skip_download": True, "noplaylist": True, "no_warnings": True} | |
| with yt.YoutubeDL(opts) as ydl: | |
| info = ydl.extract_info(url, download=False) | |
| return { | |
| "id": info.get("id"), | |
| "title": info.get("title"), | |
| "duration_sec": info.get("duration"), | |
| "uploader": info.get("uploader"), | |
| "webpage_url": info.get("webpage_url", url), | |
| "license": info.get("license"), | |
| } | |
| def _download(url: str, video_id: str, config: Config, clip_to_max: bool = True) -> Path: | |
| yt = _require_ytdlp() | |
| out_dir = work_dir(video_id) | |
| out_dir.mkdir(parents=True, exist_ok=True) | |
| opts: Dict[str, object] = { | |
| "quiet": True, | |
| "no_warnings": True, | |
| "noprogress": True, | |
| "noplaylist": True, | |
| "restrictfilenames": True, | |
| "outtmpl": str(out_dir / "source.%(ext)s"), | |
| # Keep it light: prefer <=720p, merge to mp4 when needed. | |
| "format": "bestvideo[height<=720]+bestaudio/best[height<=720]/best", | |
| "merge_output_format": "mp4", | |
| } | |
| # Duration guard -> clip to the first MAX_VIDEO_DURATION_SEC seconds. | |
| info = probe_url(url, config) | |
| duration = info.get("duration_sec") or 0 | |
| max_dur = config.max_video_duration_sec | |
| if max_dur and duration and duration > max_dur: | |
| if clip_to_max: | |
| warnings.warn( | |
| f"Video is {int(duration)}s (> MAX_VIDEO_DURATION_SEC={max_dur}); " | |
| f"downloading only the first {max_dur}s." | |
| ) | |
| opts["download_ranges"] = yt.utils.download_range_func(None, [(0, max_dur)]) | |
| opts["force_keyframes_at_cuts"] = True | |
| else: | |
| raise ValueError( | |
| f"Video is {int(duration)}s, which exceeds " | |
| f"MAX_VIDEO_DURATION_SEC={max_dur}." | |
| ) | |
| with yt.YoutubeDL(opts) as ydl: | |
| ydl.download([url]) | |
| candidates = [ | |
| p for p in out_dir.glob("source.*") if p.suffix.lower() in _MEDIA_EXTS | |
| ] | |
| if not candidates: | |
| raise RuntimeError(f"Download produced no media file for {url}") | |
| # Largest file is the merged/best media. | |
| return max(candidates, key=lambda p: p.stat().st_size) | |
| def resolve_source( | |
| source: str, | |
| video_id: str = None, | |
| config: Config = CONFIG, | |
| ) -> Tuple[Path, Dict[str, object]]: | |
| """Return ``(local_video_path, source_info)`` for a path or URL. | |
| ``source_info`` always carries a suggested ``video_id`` plus provenance the | |
| pipeline records into the metrics artifact. | |
| """ | |
| if is_url(source): | |
| meta = probe_url(source, config) | |
| vid = video_id or slugify(str(meta.get("id") or meta.get("title") or "video")) | |
| path = _download(source, vid, config) | |
| meta.update({"video_id": vid, "source": "url", "source_url": source}) | |
| return path, meta | |
| path = Path(source) | |
| if not path.exists(): | |
| raise FileNotFoundError(f"Video not found: {path}") | |
| return path, { | |
| "video_id": video_id or slugify(path.stem), | |
| "title": path.stem, | |
| "source": "local", | |
| "source_path": str(path), | |
| } | |