File size: 6,210 Bytes
bcfee56
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4cf533e
 
 
 
 
bcfee56
 
 
 
 
4cf533e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
bcfee56
 
 
 
 
 
 
 
 
 
 
 
 
 
4cf533e
 
 
 
bcfee56
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4cf533e
 
bcfee56
4cf533e
bcfee56
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
"""Download a YouTube video via the ``youtube-video-fast-downloader-24-7`` RapidAPI.

The API returns a link on the provider's OWN server (e.g. ``s5-audio.12388101.xyz``), not
a ``googlevideo.com`` URL — so the Space downloads the file directly, bypassing both the
egress DPI and the datacenter-IP block, with no IP-lock. The link is prepared
asynchronously: it 404s for ~20-300s while the provider fetches it, then is live for about
10 minutes. We poll until it's ready, then stream it to disk.

Requires the ``RAPIDAPI_KEY`` Space secret. Quality ``18`` is 360p muxed (video+audio) —
small, and its audio track is enough for faster-whisper.
"""
from __future__ import annotations

import os
import time

import requests

RAPIDAPI_HOST = "youtube-video-fast-downloader-24-7.p.rapidapi.com"
DEFAULT_QUALITY = "18"  # itag 18 = 360p muxed (has audio, ~10-30 MB for typical videos)

# RapidAPI tracks monthly usage against the plan limit and returns it in every response
# header. We mirror the latest values here and enforce the cap — no persistence needed,
# and it's Space-wide + monthly (the plan's billing cycle) by construction.
_quota = {"limit": None, "remaining": None, "reset": None}  # reset = seconds until reset


class DownloadError(RuntimeError):
    """Raised when the video can't be obtained from the download API."""


def _update_quota(headers) -> None:
    def _int(name):
        try:
            return int(headers.get(name))
        except (TypeError, ValueError):
            return None
    for key, hdr in (("limit", "X-RateLimit-Requests-Limit"),
                     ("remaining", "X-RateLimit-Requests-Remaining"),
                     ("reset", "X-RateLimit-Requests-Reset")):
        val = _int(hdr)
        if val is not None:
            _quota[key] = val


def quota_status() -> dict:
    """Latest known monthly quota: ``{limit, remaining, reset(sec), used, self_cap}``."""
    q = dict(_quota)
    q["used"] = (q["limit"] - q["remaining"]) if (q["limit"] and q["remaining"] is not None) else None
    q["self_cap"] = _self_cap()
    return q


def _reset_days() -> str:
    r = _quota.get("reset")
    return f"~{max(1, r // 86400)} day(s)" if r else "the next cycle"


def _self_cap() -> int | None:
    """Optional lower monthly cap (env ``MONTHLY_DOWNLOAD_LIMIT``); None = use plan limit."""
    v = os.environ.get("MONTHLY_DOWNLOAD_LIMIT", "").strip()
    try:
        return int(v) if v else None
    except ValueError:
        return None


def _enforce_quota() -> None:
    """Refuse before spending a request if the monthly cap is already reached."""
    q = _quota
    if q["remaining"] is None:
        return  # unknown yet (fresh start) — let the call itself surface a 429
    if q["limit"] is not None:
        used = q["limit"] - q["remaining"]
        cap = _self_cap()
        if cap is not None and used >= cap:
            raise DownloadError(
                f"Self-imposed monthly cap reached: {used} of {cap} used. Resets in {_reset_days()}.")
    if q["remaining"] <= 0:
        raise DownloadError(
            f"Monthly request limit reached (plan: {q['limit']}). Resets in {_reset_days()}.")


def _headers() -> dict:
    key = os.environ.get("RAPIDAPI_KEY", "").strip()
    if not key:
        raise DownloadError("RAPIDAPI_KEY is not set (required for the video download API).")
    return {"X-RapidAPI-Key": key, "X-RapidAPI-Host": RAPIDAPI_HOST}


def _request_urls(video_id: str, quality: str, timeout: int = 60) -> tuple[list[str], dict]:
    """Ask the API for a download link; return candidate URLs (primary + reserved)."""
    url = f"https://{RAPIDAPI_HOST}/download_video/{video_id}"
    try:
        r = requests.get(url, params={"quality": quality}, headers=_headers(), timeout=timeout)
    except requests.RequestException as exc:
        raise DownloadError(f"download API unreachable: {exc}") from exc
    _update_quota(r.headers)
    if r.status_code == 429:
        raise DownloadError(f"Monthly request limit reached (plan: {_quota.get('limit')}). "
                            f"Resets in {_reset_days()}.")
    if r.status_code in (401, 403):
        raise DownloadError(f"download API auth failed (HTTP {r.status_code}); "
                            "check RAPIDAPI_KEY / that you're subscribed.")
    if r.status_code != 200:
        raise DownloadError(f"download API HTTP {r.status_code}: {r.text[:160]}")
    data = r.json()
    urls = [u for u in (data.get("file"), data.get("reserved_file")) if u]
    if not urls:
        raise DownloadError(f"no download URL in API response: {str(data)[:200]}")
    return urls, data


def download_video(video_id: str, dest: str, quality: str = DEFAULT_QUALITY,
                   poll_timeout: int = 330, chunk: int = 1 << 20,
                   progress=None) -> str:
    """Download ``video_id`` to ``dest`` and return the path.

    Polls the (async) provider link until ready (404 -> wait), then streams it to ``dest``.
    Raises DownloadError if it never becomes ready within ``poll_timeout`` seconds, or if
    the monthly request cap is already reached.
    """
    _enforce_quota()  # refuse before spending a request if the monthly cap is reached
    urls, _ = _request_urls(video_id, quality)
    deadline = time.time() + poll_timeout
    last = "not ready"
    while time.time() < deadline:
        for url in urls:
            try:
                with requests.get(url, stream=True, timeout=90) as resp:
                    if resp.status_code == 404:
                        last = "404 (server still preparing)"
                        continue
                    resp.raise_for_status()
                    with open(dest, "wb") as fh:
                        for c in resp.iter_content(chunk):
                            if c:
                                fh.write(c)
                if os.path.getsize(dest) > 0:
                    return dest
            except requests.RequestException as exc:
                last = f"{type(exc).__name__}: {exc}"
        if progress:
            progress(0.0, desc="Preparing video (server-side, up to ~5 min)…")
        time.sleep(8)
    raise DownloadError(f"video not ready after {poll_timeout}s ({last}).")