TutorialMaker / app.py
vivekchakraverty's picture
Add PO-token support (manual PO token + visitor data) with README guide
dc57844
Raw
History Blame
16.4 kB
"""Gradio Space: YouTube topic -> captioned .docx tutorial.
Orchestrates the pipeline stages and streams progress/status to the UI. Heavy ML
imports (torch/transformers/faster-whisper) are lazy inside the pipeline modules, so app
startup stays fast.
"""
from __future__ import annotations
import base64
import binascii
import os
import re
import shutil
import tempfile
import gradio as gr
from pipeline import (
captions as captions_mod,
docx_builder,
download as download_mod,
frames as frames_mod,
search as search_mod,
sentiment as sentiment_mod,
transcribe as transcribe_mod,
tutorial as tutorial_mod,
)
LLM_CHOICES = [
"deepseek-ai/DeepSeek-V3",
"meta-llama/Llama-3.3-70B-Instruct",
"openai/gpt-oss-120b",
]
VLM_CHOICES = [
"Qwen/Qwen2.5-VL-72B-Instruct",
"Qwen/Qwen2.5-VL-7B-Instruct",
"meta-llama/Llama-3.2-90B-Vision-Instruct",
]
def _looks_like_netscape(text: str) -> bool:
"""True if ``text`` is already a tab-separated Netscape cookie file."""
head = text.lstrip()
return head.startswith("#") or "\tTRUE\t" in text or "\tFALSE\t" in text
def _maybe_b64_decode(text: str) -> str | None:
"""If ``text`` is base64 that decodes to a cookie file, return the decoded text.
HF secret fields can turn the required TAB characters into spaces, which breaks the
Netscape format. Pasting base64 of the file avoids that; we auto-detect and decode it.
"""
compact = "".join(text.split())
if len(compact) < 16 or re.search(r"[^A-Za-z0-9+/=]", compact):
return None
try:
decoded = base64.b64decode(compact, validate=True).decode("utf-8", "replace")
except (binascii.Error, ValueError):
return None
return decoded if _looks_like_netscape(decoded) else None
def _cookiefile(workdir: str, raw: str | None = None) -> str | None:
"""Materialize cookies to a Netscape cookie file on disk; return its path or None.
``raw`` is the per-user UI value; if empty we fall back to the operator-wide
``YT_COOKIES`` secret. Accepts either raw cookies.txt contents (tabs preserved) or a
base64 encoding of them. A missing header line is added so yt-dlp accepts the file.
"""
data = (raw or "").strip() or os.environ.get("YT_COOKIES")
if not data or not data.strip():
return None
if not _looks_like_netscape(data):
decoded = _maybe_b64_decode(data)
if decoded:
data = decoded
if not data.lstrip().startswith(("# Netscape", "# HTTP")):
data = "# Netscape HTTP Cookie File\n" + data.lstrip("\n")
if not data.endswith("\n"):
data += "\n"
path = os.path.join(workdir, "cookies.txt")
with open(path, "w", encoding="utf-8", newline="\n") as fh:
fh.write(data)
return path
def _resolve_proxy(raw: str | None = None) -> str | None:
"""Per-user proxy URL, falling back to the operator-wide YT_PROXY secret."""
proxy = (raw or "").strip() or os.environ.get("YT_PROXY", "").strip()
return proxy or None
def _resolve_pot(po_token: str | None, visitor_data: str | None) -> tuple[str | None, str | None]:
"""Per-user PO token + visitor data, falling back to YT_POT / YT_VISITOR_DATA secrets."""
pot = (po_token or "").strip() or os.environ.get("YT_POT", "").strip()
vis = (visitor_data or "").strip() or os.environ.get("YT_VISITOR_DATA", "").strip()
return (pot or None), (vis or None)
def _ranking_rows(scored: list[dict]) -> list[list]:
rows = []
for rank, v in enumerate(scored, start=1):
rows.append([
rank,
v.get("title", v["video_id"]),
f"{v['positive_share'] * 100:.0f}%",
v.get("n_comments", 0),
v.get("note", "") or "ok",
v["url"],
])
return rows
def _safe_name(text: str) -> str:
return re.sub(r"[^A-Za-z0-9._-]+", "_", text).strip("_")[:60] or "tutorial"
def _collect_keywords(primary_kw, secondary_kw) -> dict:
"""Build ``{"primary": str, "secondary": [str, ...]}`` from the two keyword inputs.
Secondary keywords are comma-separated. Duplicates and the primary are removed.
"""
primary = (primary_kw or "").strip()
secondary = []
seen = {primary.lower()}
for part in (secondary_kw or "").split(","):
kw = part.strip()
if kw and kw.lower() not in seen:
seen.add(kw.lower())
secondary.append(kw)
return {"primary": primary, "secondary": secondary}
def run_pipeline(topic, hf_token, llm_model, vlm_model, w_llm, w_whisper, lead,
max_minutes, max_shots, primary_kw, secondary_kw,
cookies_text, proxy_url, po_token_in, visitor_data_in,
progress=gr.Progress()):
"""Generator that yields (status_md, ranking_df, transcript, docx_file)."""
log: list[str] = []
def status(msg: str):
log.append(msg)
return "\n\n".join(log)
topic = (topic or "").strip()
if not topic:
raise gr.Error("Please enter a topic.")
if not (hf_token or "").strip():
raise gr.Error("Please paste your Hugging Face token (used for the LLM + vision model).")
workdir = tempfile.mkdtemp(prefix="ytt_")
frames_dir = os.path.join(workdir, "frames")
video_path = None
try:
cookiefile = _cookiefile(workdir, cookies_text)
proxy = _resolve_proxy(proxy_url)
po_token, visitor_data = _resolve_pot(po_token_in, visitor_data_in)
auth_bits = []
if cookiefile:
auth_bits.append("cookies")
if proxy:
auth_bits.append("proxy")
if po_token:
auth_bits.append("PO token")
if auth_bits:
yield status("🔐 Using " + " + ".join(auth_bits) + " for YouTube access."), gr.update(), gr.update(), gr.update()
# 1. Search ------------------------------------------------------------------
progress(0.02, desc="Searching")
yield status(f"🔍 Searching top videos for **{topic}**…"), gr.update(), gr.update(), gr.update()
videos = search_mod.search_top5(topic)
yield status(f"Found {len(videos)} candidate videos."), gr.update(), gr.update(), gr.update()
# 2. Sentiment ranking -------------------------------------------------------
yield status("💬 Fetching comments and scoring sentiment…"), gr.update(), gr.update(), gr.update()
best, scored = sentiment_mod.rank_by_sentiment(
videos, cookiefile, progress, proxy, po_token, visitor_data)
ranking = gr.update(value=_ranking_rows(scored))
yield (status(f"🏆 Picked **{best.get('title', best['video_id'])}** "
f"({best['positive_share'] * 100:.0f}% positive)."),
ranking, gr.update(), gr.update())
# 3. Download + audio --------------------------------------------------------
progress(0.25, desc="Downloading")
yield status("⬇️ Downloading the chosen video…"), ranking, gr.update(), gr.update()
video_path, duration = download_mod.download_video(
best["url"], workdir, cookiefile, int(max_minutes), proxy, po_token, visitor_data)
wav = download_mod.extract_audio(video_path, workdir)
# 4. Transcribe --------------------------------------------------------------
progress(0.4, desc="Transcribing")
yield status("📝 Transcribing with Whisper (this is the slow part on CPU)…"), ranking, gr.update(), gr.update()
segs = transcribe_mod.transcribe(wav, progress)
transcript = transcribe_mod.transcript_text(segs)
yield (status(f"Transcript ready ({len(segs)} segments)."),
ranking, gr.update(value=transcript), gr.update())
# 5. Candidate frames, then DELETE the video --------------------------------
progress(0.6, desc="Extracting frames")
candidates = frames_mod.extract_candidates(video_path, frames_dir, duration)
frames_mod.delete_video(video_path)
video_path = None
yield (status(f"🎞️ Extracted {len(candidates)} candidate frames and "
f"**deleted the downloaded video**."),
ranking, gr.update(value=transcript), gr.update())
# 6. Tutorial text -----------------------------------------------------------
progress(0.72, desc="Writing tutorial")
keywords = _collect_keywords(primary_kw, secondary_kw)
kw_note = f" • primary: '{keywords['primary']}'" if keywords["primary"] else ""
if keywords["secondary"]:
kw_note += f" • secondary: {', '.join(keywords['secondary'])}"
yield status(f"🤖 Generating tutorial with `{llm_model}`{kw_note}…"), ranking, gr.update(value=transcript), gr.update()
tut = tutorial_mod.generate_tutorial(transcript, hf_token.strip(), llm_model, keywords)
if keywords["primary"]:
n = tutorial_mod.count_keyword(tut, keywords["primary"])
yield (status(f"🔑 Primary keyword '{keywords['primary']}' appears {n}× in the post."),
ranking, gr.update(value=transcript), gr.update())
# 7. Weighted screenshot selection ------------------------------------------
selected = frames_mod.select_screenshots(
tut["steps"], segs, candidates,
w_llm=float(w_llm), w_whisper=float(w_whisper), lead=float(lead),
max_shots=int(max_shots),
)
yield (status(f"🖼️ Selected {len(selected)} screenshots via the weighted indicator."),
ranking, gr.update(value=transcript), gr.update())
# 8. Captions ----------------------------------------------------------------
progress(0.85, desc="Captioning")
yield status(f"✍️ Captioning screenshots with `{vlm_model}`…"), ranking, gr.update(value=transcript), gr.update()
caps = captions_mod.caption_frames(selected, tut["steps"], hf_token.strip(), vlm_model, progress)
# 9. DOCX --------------------------------------------------------------------
progress(0.95, desc="Building document")
out_path = os.path.join(workdir, f"{_safe_name(tut['title'])}.docx")
docx_builder.build_docx(tut, selected, caps, out_path, source_url=best["url"])
progress(1.0, desc="Done")
yield (status("✅ Done! Download your tutorial below."),
ranking, gr.update(value=transcript), gr.update(value=out_path))
except gr.Error:
raise
except (download_mod.DownloadError, RuntimeError, ValueError) as exc:
raise gr.Error(str(exc))
finally:
# Always remove the video if it somehow survived; keep frames/docx until the
# response is sent (Gradio copies the returned file out).
if video_path:
frames_mod.delete_video(video_path)
def build_ui():
with gr.Blocks(title="YouTube → Tutorial Post") as demo:
gr.Markdown(
"# 📝 YouTube → Tutorial Post Generator\n"
"Enter a topic and your Hugging Face token. The Space picks the best video, "
"transcribes it, and builds a **captioned `.docx` tutorial**. Your token is "
"used only for the LLM + vision-model calls and **billed to your account**."
)
with gr.Row():
with gr.Column(scale=2):
topic = gr.Textbox(label="Topic", placeholder="e.g. Excel pivot tables for beginners")
hf_token = gr.Textbox(label="Hugging Face token", type="password",
placeholder="hf_… (Inference Providers permission)")
with gr.Column(scale=1):
llm_model = gr.Dropdown(LLM_CHOICES, value=LLM_CHOICES[0],
label="Tutorial LLM", allow_custom_value=True)
vlm_model = gr.Dropdown(VLM_CHOICES, value=VLM_CHOICES[0],
label="Vision model (captions)", allow_custom_value=True)
with gr.Accordion("YouTube access — cookies / proxy (often required)", open=False):
gr.Markdown(
"⚠️ **YouTube usually blocks the Space's datacenter IP.** To download, give "
"the Space **your own** access below — it is used only for your run and "
"deleted afterward.\n\n"
"- **Use a throwaway Google account, not your main one.** yt-dlp activity "
"can get an account rate-limited or flagged.\n"
"- **Cookies:** export a `youtube.com` cookies.txt (Netscape format) from a "
"logged-in throwaway account and paste it (raw or base64) below.\n"
"- **PO token (free, no proxy):** a Proof-of-Origin token + visitor data "
"can pass the bot-check from a datacenter IP. **See this Space's README → "
"\"PO token\" guide** for how to get and paste them.\n"
"- **Proxy:** a **residential** proxy works; **free *datacenter* proxies "
"(e.g. Webshare's free tier) usually do NOT** get past YouTube's block and "
"have tight bandwidth caps.\n"
"- An operator can instead set Space secrets `YT_COOKIES` / `YT_PROXY` / "
"`YT_POT` / `YT_VISITOR_DATA` as shared defaults."
)
cookies_text = gr.Textbox(
label="YouTube cookies (cookies.txt contents or base64)", lines=4,
placeholder="# Netscape HTTP Cookie File … (or a base64 blob)")
with gr.Row():
po_token_in = gr.Textbox(
label="PO token (see README) — CLIENT.CONTEXT+TOKEN", lines=2, scale=3,
placeholder="web.gvs+AbC…, web.player+XyZ…")
visitor_data_in = gr.Textbox(
label="Visitor data (pairs with the PO token)", scale=2,
placeholder="Cgt...%3D%3D")
proxy_url = gr.Textbox(
label="Proxy URL (optional)", type="password",
placeholder="http://user:pass@host:port")
with gr.Accordion("SEO / AEO keywords (optional)", open=False):
gr.Markdown(
"The **primary keyword** is used naturally ~3× in the body and placed in "
"the title, URL slug, meta description, the first 100 words, and one or "
"two H2 headings. Each **secondary keyword** is used once. The post also "
"follows answer-engine best practices (direct answer up top, FAQ, "
"last-updated date, source citation)."
)
primary_kw = gr.Textbox(label="Primary keyword",
placeholder="e.g. godot ai plugin")
secondary_kw = gr.Textbox(label="Secondary keywords (comma-separated)",
placeholder="e.g. gdscript assistant, ai game tools")
with gr.Accordion("Advanced settings", open=False):
with gr.Row():
w_llm = gr.Slider(0.0, 1.0, value=0.4, step=0.05, label="Weight: LLM timestamp")
w_whisper = gr.Slider(0.0, 1.0, value=0.6, step=0.05, label="Weight: Whisper timing")
lead = gr.Slider(0.0, 5.0, value=1.0, step=0.5, label="Lead offset (s)")
with gr.Row():
max_minutes = gr.Slider(2, 60, value=20, step=1, label="Max video length (min)")
max_shots = gr.Slider(1, 15, value=8, step=1, label="Max screenshots")
run_btn = gr.Button("Generate tutorial", variant="primary")
status_md = gr.Markdown(label="Status")
ranking_df = gr.Dataframe(
headers=["#", "Title", "Positive", "Comments", "Note", "URL"],
label="Sentiment ranking", interactive=False, wrap=True,
)
transcript_box = gr.Textbox(label="Transcript preview", lines=10, max_lines=20)
docx_file = gr.File(label="Download tutorial (.docx)")
run_btn.click(
run_pipeline,
inputs=[topic, hf_token, llm_model, vlm_model, w_llm, w_whisper, lead,
max_minutes, max_shots, primary_kw, secondary_kw,
cookies_text, proxy_url, po_token_in, visitor_data_in],
outputs=[status_md, ranking_df, transcript_box, docx_file],
)
return demo
if __name__ == "__main__":
build_ui().queue().launch()