"""Gradio Space: YouTube topic -> captioned .docx tutorial. Orchestrates the pipeline stages and streams progress/status to the UI. Heavy ML imports (torch/transformers/faster-whisper) are lazy inside the pipeline modules, so app startup stays fast. """ from __future__ import annotations import base64 import binascii import os import re import shutil import tempfile import gradio as gr from pipeline import ( captions as captions_mod, docx_builder, download as download_mod, frames as frames_mod, search as search_mod, sentiment as sentiment_mod, transcribe as transcribe_mod, tutorial as tutorial_mod, ) LLM_CHOICES = [ "deepseek-ai/DeepSeek-V3", "meta-llama/Llama-3.3-70B-Instruct", "openai/gpt-oss-120b", ] VLM_CHOICES = [ "Qwen/Qwen2.5-VL-72B-Instruct", "Qwen/Qwen2.5-VL-7B-Instruct", "meta-llama/Llama-3.2-90B-Vision-Instruct", ] def _looks_like_netscape(text: str) -> bool: """True if ``text`` is already a tab-separated Netscape cookie file.""" head = text.lstrip() return head.startswith("#") or "\tTRUE\t" in text or "\tFALSE\t" in text def _maybe_b64_decode(text: str) -> str | None: """If ``text`` is base64 that decodes to a cookie file, return the decoded text. HF secret fields can turn the required TAB characters into spaces, which breaks the Netscape format. Pasting base64 of the file avoids that; we auto-detect and decode it. """ compact = "".join(text.split()) if len(compact) < 16 or re.search(r"[^A-Za-z0-9+/=]", compact): return None try: decoded = base64.b64decode(compact, validate=True).decode("utf-8", "replace") except (binascii.Error, ValueError): return None return decoded if _looks_like_netscape(decoded) else None def _cookiefile(workdir: str, raw: str | None = None) -> str | None: """Materialize cookies to a Netscape cookie file on disk; return its path or None. ``raw`` is the per-user UI value; if empty we fall back to the operator-wide ``YT_COOKIES`` secret. Accepts either raw cookies.txt contents (tabs preserved) or a base64 encoding of them. A missing header line is added so yt-dlp accepts the file. """ data = (raw or "").strip() or os.environ.get("YT_COOKIES") if not data or not data.strip(): return None if not _looks_like_netscape(data): decoded = _maybe_b64_decode(data) if decoded: data = decoded if not data.lstrip().startswith(("# Netscape", "# HTTP")): data = "# Netscape HTTP Cookie File\n" + data.lstrip("\n") if not data.endswith("\n"): data += "\n" path = os.path.join(workdir, "cookies.txt") with open(path, "w", encoding="utf-8", newline="\n") as fh: fh.write(data) return path def _resolve_proxy(raw: str | None = None) -> str | None: """Per-user proxy URL, falling back to the operator-wide YT_PROXY secret.""" proxy = (raw or "").strip() or os.environ.get("YT_PROXY", "").strip() return proxy or None def _resolve_pot(po_token: str | None, visitor_data: str | None) -> tuple[str | None, str | None]: """Per-user PO token + visitor data, falling back to YT_POT / YT_VISITOR_DATA secrets.""" pot = (po_token or "").strip() or os.environ.get("YT_POT", "").strip() vis = (visitor_data or "").strip() or os.environ.get("YT_VISITOR_DATA", "").strip() return (pot or None), (vis or None) def _ranking_rows(scored: list[dict]) -> list[list]: rows = [] for rank, v in enumerate(scored, start=1): rows.append([ rank, v.get("title", v["video_id"]), f"{v['positive_share'] * 100:.0f}%", v.get("n_comments", 0), v.get("note", "") or "ok", v["url"], ]) return rows def _safe_name(text: str) -> str: return re.sub(r"[^A-Za-z0-9._-]+", "_", text).strip("_")[:60] or "tutorial" def _collect_keywords(primary_kw, secondary_kw) -> dict: """Build ``{"primary": str, "secondary": [str, ...]}`` from the two keyword inputs. Secondary keywords are comma-separated. Duplicates and the primary are removed. """ primary = (primary_kw or "").strip() secondary = [] seen = {primary.lower()} for part in (secondary_kw or "").split(","): kw = part.strip() if kw and kw.lower() not in seen: seen.add(kw.lower()) secondary.append(kw) return {"primary": primary, "secondary": secondary} def run_pipeline(topic, hf_token, llm_model, vlm_model, w_llm, w_whisper, lead, max_minutes, max_shots, primary_kw, secondary_kw, cookies_text, proxy_url, po_token_in, visitor_data_in, progress=gr.Progress()): """Generator that yields (status_md, ranking_df, transcript, docx_file).""" log: list[str] = [] def status(msg: str): log.append(msg) return "\n\n".join(log) topic = (topic or "").strip() if not topic: raise gr.Error("Please enter a topic.") if not (hf_token or "").strip(): raise gr.Error("Please paste your Hugging Face token (used for the LLM + vision model).") workdir = tempfile.mkdtemp(prefix="ytt_") frames_dir = os.path.join(workdir, "frames") video_path = None try: cookiefile = _cookiefile(workdir, cookies_text) proxy = _resolve_proxy(proxy_url) po_token, visitor_data = _resolve_pot(po_token_in, visitor_data_in) auth_bits = [] if cookiefile: auth_bits.append("cookies") if proxy: auth_bits.append("proxy") if po_token: auth_bits.append("PO token") if auth_bits: yield status("🔐 Using " + " + ".join(auth_bits) + " for YouTube access."), gr.update(), gr.update(), gr.update() # 1. Search ------------------------------------------------------------------ progress(0.02, desc="Searching") yield status(f"🔍 Searching top videos for **{topic}**…"), gr.update(), gr.update(), gr.update() videos = search_mod.search_top5(topic) yield status(f"Found {len(videos)} candidate videos."), gr.update(), gr.update(), gr.update() # 2. Sentiment ranking ------------------------------------------------------- yield status("💬 Fetching comments and scoring sentiment…"), gr.update(), gr.update(), gr.update() best, scored = sentiment_mod.rank_by_sentiment( videos, cookiefile, progress, proxy, po_token, visitor_data) ranking = gr.update(value=_ranking_rows(scored)) yield (status(f"🏆 Picked **{best.get('title', best['video_id'])}** " f"({best['positive_share'] * 100:.0f}% positive)."), ranking, gr.update(), gr.update()) # 3. Download + audio -------------------------------------------------------- progress(0.25, desc="Downloading") yield status("⬇️ Downloading the chosen video…"), ranking, gr.update(), gr.update() video_path, duration = download_mod.download_video( best["url"], workdir, cookiefile, int(max_minutes), proxy, po_token, visitor_data) wav = download_mod.extract_audio(video_path, workdir) # 4. Transcribe -------------------------------------------------------------- progress(0.4, desc="Transcribing") yield status("📝 Transcribing with Whisper (this is the slow part on CPU)…"), ranking, gr.update(), gr.update() segs = transcribe_mod.transcribe(wav, progress) transcript = transcribe_mod.transcript_text(segs) yield (status(f"Transcript ready ({len(segs)} segments)."), ranking, gr.update(value=transcript), gr.update()) # 5. Candidate frames, then DELETE the video -------------------------------- progress(0.6, desc="Extracting frames") candidates = frames_mod.extract_candidates(video_path, frames_dir, duration) frames_mod.delete_video(video_path) video_path = None yield (status(f"🎞️ Extracted {len(candidates)} candidate frames and " f"**deleted the downloaded video**."), ranking, gr.update(value=transcript), gr.update()) # 6. Tutorial text ----------------------------------------------------------- progress(0.72, desc="Writing tutorial") keywords = _collect_keywords(primary_kw, secondary_kw) kw_note = f" • primary: '{keywords['primary']}'" if keywords["primary"] else "" if keywords["secondary"]: kw_note += f" • secondary: {', '.join(keywords['secondary'])}" yield status(f"🤖 Generating tutorial with `{llm_model}`{kw_note}…"), ranking, gr.update(value=transcript), gr.update() tut = tutorial_mod.generate_tutorial(transcript, hf_token.strip(), llm_model, keywords) if keywords["primary"]: n = tutorial_mod.count_keyword(tut, keywords["primary"]) yield (status(f"🔑 Primary keyword '{keywords['primary']}' appears {n}× in the post."), ranking, gr.update(value=transcript), gr.update()) # 7. Weighted screenshot selection ------------------------------------------ selected = frames_mod.select_screenshots( tut["steps"], segs, candidates, w_llm=float(w_llm), w_whisper=float(w_whisper), lead=float(lead), max_shots=int(max_shots), ) yield (status(f"🖼️ Selected {len(selected)} screenshots via the weighted indicator."), ranking, gr.update(value=transcript), gr.update()) # 8. Captions ---------------------------------------------------------------- progress(0.85, desc="Captioning") yield status(f"✍️ Captioning screenshots with `{vlm_model}`…"), ranking, gr.update(value=transcript), gr.update() caps = captions_mod.caption_frames(selected, tut["steps"], hf_token.strip(), vlm_model, progress) # 9. DOCX -------------------------------------------------------------------- progress(0.95, desc="Building document") out_path = os.path.join(workdir, f"{_safe_name(tut['title'])}.docx") docx_builder.build_docx(tut, selected, caps, out_path, source_url=best["url"]) progress(1.0, desc="Done") yield (status("✅ Done! Download your tutorial below."), ranking, gr.update(value=transcript), gr.update(value=out_path)) except gr.Error: raise except (download_mod.DownloadError, RuntimeError, ValueError) as exc: raise gr.Error(str(exc)) finally: # Always remove the video if it somehow survived; keep frames/docx until the # response is sent (Gradio copies the returned file out). if video_path: frames_mod.delete_video(video_path) def build_ui(): with gr.Blocks(title="YouTube → Tutorial Post") as demo: gr.Markdown( "# 📝 YouTube → Tutorial Post Generator\n" "Enter a topic and your Hugging Face token. The Space picks the best video, " "transcribes it, and builds a **captioned `.docx` tutorial**. Your token is " "used only for the LLM + vision-model calls and **billed to your account**." ) with gr.Row(): with gr.Column(scale=2): topic = gr.Textbox(label="Topic", placeholder="e.g. Excel pivot tables for beginners") hf_token = gr.Textbox(label="Hugging Face token", type="password", placeholder="hf_… (Inference Providers permission)") with gr.Column(scale=1): llm_model = gr.Dropdown(LLM_CHOICES, value=LLM_CHOICES[0], label="Tutorial LLM", allow_custom_value=True) vlm_model = gr.Dropdown(VLM_CHOICES, value=VLM_CHOICES[0], label="Vision model (captions)", allow_custom_value=True) with gr.Accordion("YouTube access — cookies / proxy (often required)", open=False): gr.Markdown( "⚠️ **YouTube usually blocks the Space's datacenter IP.** To download, give " "the Space **your own** access below — it is used only for your run and " "deleted afterward.\n\n" "- **Use a throwaway Google account, not your main one.** yt-dlp activity " "can get an account rate-limited or flagged.\n" "- **Cookies:** export a `youtube.com` cookies.txt (Netscape format) from a " "logged-in throwaway account and paste it (raw or base64) below.\n" "- **PO token (free, no proxy):** a Proof-of-Origin token + visitor data " "can pass the bot-check from a datacenter IP. **See this Space's README → " "\"PO token\" guide** for how to get and paste them.\n" "- **Proxy:** a **residential** proxy works; **free *datacenter* proxies " "(e.g. Webshare's free tier) usually do NOT** get past YouTube's block and " "have tight bandwidth caps.\n" "- An operator can instead set Space secrets `YT_COOKIES` / `YT_PROXY` / " "`YT_POT` / `YT_VISITOR_DATA` as shared defaults." ) cookies_text = gr.Textbox( label="YouTube cookies (cookies.txt contents or base64)", lines=4, placeholder="# Netscape HTTP Cookie File … (or a base64 blob)") with gr.Row(): po_token_in = gr.Textbox( label="PO token (see README) — CLIENT.CONTEXT+TOKEN", lines=2, scale=3, elem_id="tm_pot_token", placeholder="web.gvs+AbC…, web.player+XyZ…") visitor_data_in = gr.Textbox( label="Visitor data (pairs with the PO token)", scale=2, elem_id="tm_visitor_data", placeholder="Cgt...%3D%3D") proxy_url = gr.Textbox( label="Proxy URL (optional)", type="password", placeholder="http://user:pass@host:port") with gr.Accordion("SEO / AEO keywords (optional)", open=False): gr.Markdown( "The **primary keyword** is used naturally ~3× in the body and placed in " "the title, URL slug, meta description, the first 100 words, and one or " "two H2 headings. Each **secondary keyword** is used once. The post also " "follows answer-engine best practices (direct answer up top, FAQ, " "last-updated date, source citation)." ) primary_kw = gr.Textbox(label="Primary keyword", placeholder="e.g. godot ai plugin") secondary_kw = gr.Textbox(label="Secondary keywords (comma-separated)", placeholder="e.g. gdscript assistant, ai game tools") with gr.Accordion("Advanced settings", open=False): with gr.Row(): w_llm = gr.Slider(0.0, 1.0, value=0.4, step=0.05, label="Weight: LLM timestamp") w_whisper = gr.Slider(0.0, 1.0, value=0.6, step=0.05, label="Weight: Whisper timing") lead = gr.Slider(0.0, 5.0, value=1.0, step=0.5, label="Lead offset (s)") with gr.Row(): max_minutes = gr.Slider(2, 60, value=20, step=1, label="Max video length (min)") max_shots = gr.Slider(1, 15, value=8, step=1, label="Max screenshots") run_btn = gr.Button("Generate tutorial", variant="primary") status_md = gr.Markdown(label="Status") ranking_df = gr.Dataframe( headers=["#", "Title", "Positive", "Comments", "Note", "URL"], label="Sentiment ranking", interactive=False, wrap=True, ) transcript_box = gr.Textbox(label="Transcript preview", lines=10, max_lines=20) docx_file = gr.File(label="Download tutorial (.docx)") run_btn.click( run_pipeline, inputs=[topic, hf_token, llm_model, vlm_model, w_llm, w_whisper, lead, max_minutes, max_shots, primary_kw, secondary_kw, cookies_text, proxy_url, po_token_in, visitor_data_in], outputs=[status_md, ranking_df, transcript_box, docx_file], ) return demo if __name__ == "__main__": build_ui().queue().launch()