TutorialMaker / pipeline /tutorial.py
vivekchakraverty's picture
Add YouTube -> tutorial .docx pipeline (Gradio Space)
f6a6455
Raw
History Blame
4.58 kB
"""Stage 7: transform the timestamped transcript into a structured tutorial.
Uses an HF Inference Providers chat model (default DeepSeek-V3) billed to the user's
token. The model is asked for strict JSON so the downstream weighted indicator and the
.docx builder have stable fields to work with.
"""
from __future__ import annotations
import json
import re
from huggingface_hub import InferenceClient
DEFAULT_LLM = "deepseek-ai/DeepSeek-V3"
# Rough char budget to stay clear of context limits on the free path. Long transcripts
# are truncated (with a marker); good enough for a tutorial summary.
MAX_TRANSCRIPT_CHARS = 24000
_SYSTEM = (
"You are a technical writer. You convert a timestamped video transcript into a clear, "
"step-by-step written tutorial. You ALWAYS respond with a single JSON object and no "
"prose outside it."
)
_INSTRUCTIONS = """\
Turn the transcript below into a tutorial. Return ONLY a JSON object with this schema:
{
"title": "string - concise tutorial title",
"intro": "string - 2-4 sentence overview",
"steps": [
{
"heading": "string - short step title",
"body": "string - 1-3 paragraphs explaining this step in your own words",
"quote": "string - a short, near-verbatim snippet (<=120 chars) copied from the "
"transcript line this step is based on, used to locate the moment",
"t_llm": number, // best timestamp IN SECONDS for an illustrative screenshot
"importance": number // 0..1, how worth screenshotting this step is
}
]
}
Rules:
- 4 to 10 steps. Keep quotes copied from the transcript so they can be matched back.
- t_llm must be within the transcript's time range.
- Output valid JSON only. No markdown, no comments in the actual output.
Transcript (each line is "[mm:ss] text"):
---
{transcript}
---
"""
def _extract_json(text: str) -> dict:
"""Parse the model output into a dict, tolerating code fences / stray prose."""
text = text.strip()
if text.startswith("```"):
text = re.sub(r"^```(?:json)?\s*|\s*```$", "", text, flags=re.DOTALL).strip()
try:
return json.loads(text)
except json.JSONDecodeError:
start, end = text.find("{"), text.rfind("}")
if start != -1 and end != -1 and end > start:
return json.loads(text[start:end + 1])
raise
def _normalize(data: dict) -> dict:
"""Coerce/clean fields so downstream stages never crash on missing keys."""
steps = []
for s in data.get("steps", []) or []:
try:
t = float(s.get("t_llm", 0) or 0)
except (TypeError, ValueError):
t = 0.0
try:
imp = float(s.get("importance", 0.5) or 0.5)
except (TypeError, ValueError):
imp = 0.5
steps.append({
"heading": str(s.get("heading", "Step")).strip() or "Step",
"body": str(s.get("body", "")).strip(),
"quote": str(s.get("quote", "")).strip(),
"t_llm": max(0.0, t),
"importance": min(1.0, max(0.0, imp)),
})
if not steps:
raise RuntimeError("The LLM returned no usable steps.")
return {
"title": str(data.get("title", "Tutorial")).strip() or "Tutorial",
"intro": str(data.get("intro", "")).strip(),
"steps": steps,
}
def generate_tutorial(transcript: str, hf_token: str, model: str = DEFAULT_LLM) -> dict:
"""Call the chat model and return a normalized ``{title, intro, steps}`` dict."""
if not hf_token:
raise ValueError("An HF token is required for the tutorial LLM (billed to your key).")
truncated = transcript[:MAX_TRANSCRIPT_CHARS]
if len(transcript) > MAX_TRANSCRIPT_CHARS:
truncated += "\n[... transcript truncated for length ...]"
prompt = _INSTRUCTIONS.replace("{transcript}", truncated)
client = InferenceClient(token=hf_token)
try:
resp = client.chat.completions.create(
model=model,
messages=[
{"role": "system", "content": _SYSTEM},
{"role": "user", "content": prompt},
],
temperature=0.3,
max_tokens=4000,
)
except Exception as exc:
raise RuntimeError(f"Tutorial LLM call failed ({model}): {exc}") from exc
content = resp.choices[0].message.content or ""
try:
data = _extract_json(content)
except Exception as exc:
raise RuntimeError(
f"Could not parse JSON from the LLM. First 400 chars:\n{content[:400]}"
) from exc
return _normalize(data)