ykumar2020's picture
Publish verified modular GAIA agent source
c641d5f verified
Raw
History Blame Contribute Delete
1.53 kB
"""YouTube transcript retrieval without downloading video media."""
from __future__ import annotations
import re
def video_id(url: str) -> str:
match = re.search(r"(?:v=|youtu\.be/|shorts/)([A-Za-z0-9_-]{11})", url)
if not match:
raise ValueError("Could not parse a YouTube video id")
return match.group(1)
def fetch_youtube_transcript(url: str, languages: tuple[str, ...] = ("en",)) -> str:
"""Return a timestamped YouTube transcript using the public transcript API."""
from youtube_transcript_api import YouTubeTranscriptApi
identifier = video_id(url)
api = YouTubeTranscriptApi()
fetched = api.fetch(identifier, languages=list(languages))
snippets = []
for item in fetched:
start = float(getattr(item, "start", 0.0))
text = str(getattr(item, "text", "")).replace("\n", " ").strip()
snippets.append(f"[{start:.1f}s] {text}")
if not snippets:
raise RuntimeError("YouTube returned an empty transcript")
return "\n".join(snippets)
def find_transcript_context(transcript: str, quote: str, window: int = 3) -> str:
"""Find a quoted dialogue line and return nearby timestamped transcript lines."""
lines = transcript.splitlines()
needle = re.sub(r"[^a-z0-9 ]", "", quote.lower())
for index, line in enumerate(lines):
haystack = re.sub(r"[^a-z0-9 ]", "", line.lower())
if needle and needle in haystack:
return "\n".join(lines[max(0, index - window) : index + window + 1])
return transcript