Spaces:
Sleeping
Sleeping
File size: 7,351 Bytes
23f7291 f6a6455 23f7291 f6a6455 23f7291 ffdac8f 23f7291 ffdac8f 23f7291 12e4183 f6a6455 f07812e f6a6455 23f7291 ffdac8f f07812e ffdac8f f07812e f6a6455 23f7291 f07812e 23f7291 ffdac8f 23f7291 ffdac8f 23f7291 ffdac8f 23f7291 f6a6455 f07812e ffdac8f f07812e f6a6455 f07812e f6a6455 f07812e 23f7291 f07812e ffdac8f f07812e ffdac8f f07812e | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 | """Stage 4: fetch a timestamped transcript (no video download).
Primary source is **transcriptapi.com** (a hosted transcript API). Because it's called on
its own domain — not ``youtube.com`` — the Space reaches it directly, so transcripts need
**no proxy** at all. Set the ``TRANSCRIPTAPI_KEY`` Space secret to enable it.
Fallback is **youtube-transcript-api**, which scrapes ``www.youtube.com`` and therefore
does need ``YT_PROXY`` from a datacenter IP (the block usually shows up as a TLS/SSL reset).
Either way the snippets carry ``start``/``duration`` timing, normalized into
``{start, end, text}`` segments the rest of the pipeline expects.
"""
from __future__ import annotations
import json
import os
import time
import urllib.error
import urllib.parse
import urllib.request
TRANSCRIPTAPI_URL = "https://transcriptapi.com/api/v2/youtube/transcript"
class TranscriptError(RuntimeError):
"""Raised when no usable transcript can be fetched (disabled, blocked, none)."""
def _fetch_transcriptapi(video_id: str, langs: list[str]) -> list[dict]:
"""Fetch timestamped snippets from transcriptapi.com. Requires TRANSCRIPTAPI_KEY.
Returns ``[{text, start, duration}]``. Raises on auth/credit/HTTP errors so the
caller can fall back to youtube-transcript-api.
"""
key = os.environ.get("TRANSCRIPTAPI_KEY", "").strip()
if not key:
raise RuntimeError("TRANSCRIPTAPI_KEY not set")
params = {"video_url": video_id, "format": "json", "include_timestamp": "true"}
if langs:
params["language"] = ",".join(langs)
url = TRANSCRIPTAPI_URL + "?" + urllib.parse.urlencode(params)
req = urllib.request.Request(url, headers={"Authorization": f"Bearer {key}",
"Accept": "application/json"})
try:
with urllib.request.urlopen(req, timeout=30) as resp:
data = json.load(resp)
except urllib.error.HTTPError as exc:
body = exc.read().decode("utf-8", "ignore")
if exc.code in (401, 403):
raise RuntimeError(f"transcriptapi auth failed (HTTP {exc.code}); "
f"check TRANSCRIPTAPI_KEY. {body[:120]}") from exc
if exc.code in (402, 429):
raise RuntimeError(f"transcriptapi out of credits / rate-limited "
f"(HTTP {exc.code}). {body[:120]}") from exc
raise RuntimeError(f"transcriptapi HTTP {exc.code}: {body[:160]}") from exc
snippets = data.get("transcript") or []
return [{"text": s.get("text", ""),
"start": s.get("start", 0.0),
"duration": s.get("duration", 0.0)} for s in snippets]
def _fetch_raw(video_id: str, langs: list[str], proxy: str | None) -> list[dict]:
from youtube_transcript_api import YouTubeTranscriptApi
# --- 1.x instance API ---
api = None
if proxy:
try:
from youtube_transcript_api.proxies import GenericProxyConfig
api = YouTubeTranscriptApi(
proxy_config=GenericProxyConfig(http_url=proxy, https_url=proxy))
except Exception:
api = YouTubeTranscriptApi()
else:
api = YouTubeTranscriptApi()
if hasattr(api, "fetch"):
fetched = api.fetch(video_id, languages=langs)
return [{"text": s.text, "start": s.start, "duration": s.duration} for s in fetched]
# --- 0.6.x classmethod API ---
kwargs = {"languages": langs}
if proxy:
kwargs["proxies"] = {"http": proxy, "https": proxy}
data = YouTubeTranscriptApi.get_transcript(video_id, **kwargs)
return [{"text": d["text"], "start": d["start"], "duration": d.get("duration", 0)} for d in data]
def _is_transient(exc: Exception) -> bool:
s = (type(exc).__name__ + " " + str(exc)).lower()
return any(k in s for k in ("ssl", "eof", "timed out", "timeout", "reset",
"connection", "max retries", "temporarily"))
def get_segments(video_id: str, languages=("en", "en-US", "en-GB"),
proxy: str | None = None, attempts: int = 3) -> list[dict]:
"""Return timestamped segments ``[{start, end, text}]`` for ``video_id``.
Tries transcriptapi.com first (if ``TRANSCRIPTAPI_KEY`` is set — no proxy needed),
then falls back to youtube-transcript-api (which uses ``proxy`` to dodge datacenter
blocks). Retries transient connection/TLS failures; raises TranscriptError otherwise.
"""
langs = list(languages)
raw, last, api_err = None, None, None
# Primary: transcriptapi.com (own domain -> reachable from the Space without a proxy).
if os.environ.get("TRANSCRIPTAPI_KEY", "").strip():
try:
raw = _fetch_transcriptapi(video_id, langs)
except Exception as exc:
api_err = exc # remember, but still try the fallback below
# Fallback: youtube-transcript-api (needs YT_PROXY from a datacenter IP).
if raw is None:
for attempt in range(attempts):
try:
raw = _fetch_raw(video_id, langs, proxy)
break
except Exception as exc:
last = exc
if _is_transient(exc) and attempt < attempts - 1:
time.sleep(1.5 * (attempt + 1))
continue
break
if raw is None:
hint = _hint(last) if last else "Unknown transcript error."
if api_err is not None:
hint = f"transcriptapi.com: {api_err} | fallback {hint}"
raise TranscriptError(hint)
segs = []
for r in raw:
text = (r.get("text") or "").strip()
if not text:
continue
start = float(r.get("start") or 0.0)
dur = float(r.get("duration") or 0.0)
segs.append({"start": start, "end": start + dur, "text": text})
if not segs:
raise TranscriptError("The transcript came back empty for this video.")
return segs
def transcript_text(segs: list[dict]) -> str:
"""Render segments as ``[mm:ss] text`` lines for the LLM and the UI preview."""
lines = []
for s in segs:
m, sec = divmod(int(s["start"]), 60)
lines.append(f"[{m:02d}:{sec:02d}] {s['text']}")
return "\n".join(lines)
def _hint(exc: Exception) -> str:
name = type(exc).__name__
msg = str(exc)
low = (name + " " + msg).lower()
# Check specific transcript states first (note: "transcripts" contains "ip").
if "disabled" in low or "transcriptsdisabled" in low:
return "This video has transcripts/captions disabled — pick another video."
if "notranscript" in low or "no transcript" in low:
return "No transcript is available for this video in the requested languages."
if "unavailable" in low or "videounavailable" in low:
return "The video is unavailable (private/removed/region-locked)."
if any(k in low for k in ("ssl", "eof", "reset", "connection", "max retries",
"timed out", "blocked", "forbidden", "too many", "429")):
return ("YouTube blocked the transcript request from this Space's IP "
"(datacenter IPs are commonly blocked — seen here as a TLS/SSL reset). "
"Set a residential proxy as the YT_PROXY Space secret and retry. "
f"[{name}]")
return f"Could not fetch transcript: {name}: {msg[:200]}"
|