"""Gradio Space: YouTube topic -> captioned .docx tutorial. Orchestrates the pipeline stages and streams progress/status to the UI. Heavy ML imports (torch/transformers/faster-whisper) are lazy inside the pipeline modules, so app startup stays fast. """ from __future__ import annotations import os import re import shutil import tempfile import gradio as gr from pipeline import ( captions as captions_mod, docx_builder, download as download_mod, frames as frames_mod, search as search_mod, sentiment as sentiment_mod, transcribe as transcribe_mod, tutorial as tutorial_mod, ) LLM_CHOICES = [ "deepseek-ai/DeepSeek-V3", "meta-llama/Llama-3.3-70B-Instruct", "openai/gpt-oss-120b", ] VLM_CHOICES = [ "Qwen/Qwen2.5-VL-72B-Instruct", "Qwen/Qwen2.5-VL-7B-Instruct", "meta-llama/Llama-3.2-90B-Vision-Instruct", ] def _cookiefile(workdir: str) -> str | None: """Materialize the YT_COOKIES secret (Netscape cookie file contents) to disk.""" data = os.environ.get("YT_COOKIES") if not data: return None path = os.path.join(workdir, "cookies.txt") with open(path, "w", encoding="utf-8") as fh: fh.write(data) return path def _ranking_rows(scored: list[dict]) -> list[list]: rows = [] for rank, v in enumerate(scored, start=1): rows.append([ rank, v.get("title", v["video_id"]), f"{v['positive_share'] * 100:.0f}%", v.get("n_comments", 0), v.get("note", "") or "ok", v["url"], ]) return rows def _safe_name(text: str) -> str: return re.sub(r"[^A-Za-z0-9._-]+", "_", text).strip("_")[:60] or "tutorial" def _collect_keywords(primary_kw, secondary_kw) -> dict: """Build ``{"primary": str, "secondary": [str, ...]}`` from the two keyword inputs. Secondary keywords are comma-separated. Duplicates and the primary are removed. """ primary = (primary_kw or "").strip() secondary = [] seen = {primary.lower()} for part in (secondary_kw or "").split(","): kw = part.strip() if kw and kw.lower() not in seen: seen.add(kw.lower()) secondary.append(kw) return {"primary": primary, "secondary": secondary} def run_pipeline(topic, hf_token, llm_model, vlm_model, w_llm, w_whisper, lead, max_minutes, max_shots, primary_kw, secondary_kw, progress=gr.Progress()): """Generator that yields (status_md, ranking_df, transcript, docx_file).""" log: list[str] = [] def status(msg: str): log.append(msg) return "\n\n".join(log) topic = (topic or "").strip() if not topic: raise gr.Error("Please enter a topic.") if not (hf_token or "").strip(): raise gr.Error("Please paste your Hugging Face token (used for the LLM + vision model).") workdir = tempfile.mkdtemp(prefix="ytt_") frames_dir = os.path.join(workdir, "frames") video_path = None try: cookiefile = _cookiefile(workdir) # 1. Search ------------------------------------------------------------------ progress(0.02, desc="Searching") yield status(f"🔍 Searching top videos for **{topic}**…"), gr.update(), gr.update(), gr.update() videos = search_mod.search_top5(topic) yield status(f"Found {len(videos)} candidate videos."), gr.update(), gr.update(), gr.update() # 2. Sentiment ranking ------------------------------------------------------- yield status("💬 Fetching comments and scoring sentiment…"), gr.update(), gr.update(), gr.update() best, scored = sentiment_mod.rank_by_sentiment(videos, cookiefile, progress) ranking = gr.update(value=_ranking_rows(scored)) yield (status(f"🏆 Picked **{best.get('title', best['video_id'])}** " f"({best['positive_share'] * 100:.0f}% positive)."), ranking, gr.update(), gr.update()) # 3. Download + audio -------------------------------------------------------- progress(0.25, desc="Downloading") yield status("⬇️ Downloading the chosen video…"), ranking, gr.update(), gr.update() video_path, duration = download_mod.download_video( best["url"], workdir, cookiefile, int(max_minutes)) wav = download_mod.extract_audio(video_path, workdir) # 4. Transcribe -------------------------------------------------------------- progress(0.4, desc="Transcribing") yield status("📝 Transcribing with Whisper (this is the slow part on CPU)…"), ranking, gr.update(), gr.update() segs = transcribe_mod.transcribe(wav, progress) transcript = transcribe_mod.transcript_text(segs) yield (status(f"Transcript ready ({len(segs)} segments)."), ranking, gr.update(value=transcript), gr.update()) # 5. Candidate frames, then DELETE the video -------------------------------- progress(0.6, desc="Extracting frames") candidates = frames_mod.extract_candidates(video_path, frames_dir, duration) frames_mod.delete_video(video_path) video_path = None yield (status(f"🎞️ Extracted {len(candidates)} candidate frames and " f"**deleted the downloaded video**."), ranking, gr.update(value=transcript), gr.update()) # 6. Tutorial text ----------------------------------------------------------- progress(0.72, desc="Writing tutorial") keywords = _collect_keywords(primary_kw, secondary_kw) kw_note = f" • primary: '{keywords['primary']}'" if keywords["primary"] else "" if keywords["secondary"]: kw_note += f" • secondary: {', '.join(keywords['secondary'])}" yield status(f"🤖 Generating tutorial with `{llm_model}`{kw_note}…"), ranking, gr.update(value=transcript), gr.update() tut = tutorial_mod.generate_tutorial(transcript, hf_token.strip(), llm_model, keywords) if keywords["primary"]: n = tutorial_mod.count_keyword(tut, keywords["primary"]) yield (status(f"🔑 Primary keyword '{keywords['primary']}' appears {n}× in the post."), ranking, gr.update(value=transcript), gr.update()) # 7. Weighted screenshot selection ------------------------------------------ selected = frames_mod.select_screenshots( tut["steps"], segs, candidates, w_llm=float(w_llm), w_whisper=float(w_whisper), lead=float(lead), max_shots=int(max_shots), ) yield (status(f"🖼️ Selected {len(selected)} screenshots via the weighted indicator."), ranking, gr.update(value=transcript), gr.update()) # 8. Captions ---------------------------------------------------------------- progress(0.85, desc="Captioning") yield status(f"✍️ Captioning screenshots with `{vlm_model}`…"), ranking, gr.update(value=transcript), gr.update() caps = captions_mod.caption_frames(selected, tut["steps"], hf_token.strip(), vlm_model, progress) # 9. DOCX -------------------------------------------------------------------- progress(0.95, desc="Building document") out_path = os.path.join(workdir, f"{_safe_name(tut['title'])}.docx") docx_builder.build_docx(tut, selected, caps, out_path, source_url=best["url"]) progress(1.0, desc="Done") yield (status("✅ Done! Download your tutorial below."), ranking, gr.update(value=transcript), gr.update(value=out_path)) except gr.Error: raise except (download_mod.DownloadError, RuntimeError, ValueError) as exc: raise gr.Error(str(exc)) finally: # Always remove the video if it somehow survived; keep frames/docx until the # response is sent (Gradio copies the returned file out). if video_path: frames_mod.delete_video(video_path) def build_ui(): with gr.Blocks(title="YouTube → Tutorial Post") as demo: gr.Markdown( "# 📝 YouTube → Tutorial Post Generator\n" "Enter a topic and your Hugging Face token. The Space picks the best video, " "transcribes it, and builds a **captioned `.docx` tutorial**. Your token is " "used only for the LLM + vision-model calls and **billed to your account**." ) with gr.Row(): with gr.Column(scale=2): topic = gr.Textbox(label="Topic", placeholder="e.g. Excel pivot tables for beginners") hf_token = gr.Textbox(label="Hugging Face token", type="password", placeholder="hf_… (Inference Providers permission)") with gr.Column(scale=1): llm_model = gr.Dropdown(LLM_CHOICES, value=LLM_CHOICES[0], label="Tutorial LLM", allow_custom_value=True) vlm_model = gr.Dropdown(VLM_CHOICES, value=VLM_CHOICES[0], label="Vision model (captions)", allow_custom_value=True) with gr.Accordion("SEO / AEO keywords (optional)", open=False): gr.Markdown( "The **primary keyword** is used naturally ~3× in the body and placed in " "the title, URL slug, meta description, the first 100 words, and one or " "two H2 headings. Each **secondary keyword** is used once. The post also " "follows answer-engine best practices (direct answer up top, FAQ, " "last-updated date, source citation)." ) primary_kw = gr.Textbox(label="Primary keyword", placeholder="e.g. godot ai plugin") secondary_kw = gr.Textbox(label="Secondary keywords (comma-separated)", placeholder="e.g. gdscript assistant, ai game tools") with gr.Accordion("Advanced settings", open=False): with gr.Row(): w_llm = gr.Slider(0.0, 1.0, value=0.4, step=0.05, label="Weight: LLM timestamp") w_whisper = gr.Slider(0.0, 1.0, value=0.6, step=0.05, label="Weight: Whisper timing") lead = gr.Slider(0.0, 5.0, value=1.0, step=0.5, label="Lead offset (s)") with gr.Row(): max_minutes = gr.Slider(2, 60, value=20, step=1, label="Max video length (min)") max_shots = gr.Slider(1, 15, value=8, step=1, label="Max screenshots") run_btn = gr.Button("Generate tutorial", variant="primary") status_md = gr.Markdown(label="Status") ranking_df = gr.Dataframe( headers=["#", "Title", "Positive", "Comments", "Note", "URL"], label="Sentiment ranking", interactive=False, wrap=True, ) transcript_box = gr.Textbox(label="Transcript preview", lines=10, max_lines=20) docx_file = gr.File(label="Download tutorial (.docx)") run_btn.click( run_pipeline, inputs=[topic, hf_token, llm_model, vlm_model, w_llm, w_whisper, lead, max_minutes, max_shots, primary_kw, secondary_kw], outputs=[status_md, ranking_df, transcript_box, docx_file], ) return demo if __name__ == "__main__": build_ui().queue().launch()