"""Stage 7: transform the timestamped transcript into a structured tutorial. Uses an HF Inference Providers chat model (default DeepSeek-V3) billed to the user's token. The model is asked for strict JSON so the downstream weighted indicator and the .docx builder have stable fields to work with. """ from __future__ import annotations import json import re from huggingface_hub import InferenceClient DEFAULT_LLM = "deepseek-ai/DeepSeek-V3" # Rough char budget to stay clear of context limits on the free path. Long transcripts # are truncated (with a marker); good enough for a tutorial summary. MAX_TRANSCRIPT_CHARS = 24000 _SYSTEM = ( "You are a technical writer. You convert a timestamped video transcript into a clear, " "step-by-step written tutorial. You ALWAYS respond with a single JSON object and no " "prose outside it." ) _INSTRUCTIONS = """\ Turn the transcript below into a tutorial. Return ONLY a JSON object with this schema: { "title": "string - concise tutorial title", "intro": "string - 2-4 sentence overview", "steps": [ { "heading": "string - short step title", "body": "string - 1-3 paragraphs explaining this step in your own words", "quote": "string - a short, near-verbatim snippet (<=120 chars) copied from the " "transcript line this step is based on, used to locate the moment", "t_llm": number, // best timestamp IN SECONDS for an illustrative screenshot "importance": number // 0..1, how worth screenshotting this step is } ] } Rules: - 4 to 10 steps. Keep quotes copied from the transcript so they can be matched back. - t_llm must be within the transcript's time range. - Output valid JSON only. No markdown, no comments in the actual output. Transcript (each line is "[mm:ss] text"): --- {transcript} --- """ def _extract_json(text: str) -> dict: """Parse the model output into a dict, tolerating code fences / stray prose.""" text = text.strip() if text.startswith("```"): text = re.sub(r"^```(?:json)?\s*|\s*```$", "", text, flags=re.DOTALL).strip() try: return json.loads(text) except json.JSONDecodeError: start, end = text.find("{"), text.rfind("}") if start != -1 and end != -1 and end > start: return json.loads(text[start:end + 1]) raise def _normalize(data: dict) -> dict: """Coerce/clean fields so downstream stages never crash on missing keys.""" steps = [] for s in data.get("steps", []) or []: try: t = float(s.get("t_llm", 0) or 0) except (TypeError, ValueError): t = 0.0 try: imp = float(s.get("importance", 0.5) or 0.5) except (TypeError, ValueError): imp = 0.5 steps.append({ "heading": str(s.get("heading", "Step")).strip() or "Step", "body": str(s.get("body", "")).strip(), "quote": str(s.get("quote", "")).strip(), "t_llm": max(0.0, t), "importance": min(1.0, max(0.0, imp)), }) if not steps: raise RuntimeError("The LLM returned no usable steps.") return { "title": str(data.get("title", "Tutorial")).strip() or "Tutorial", "intro": str(data.get("intro", "")).strip(), "steps": steps, } def generate_tutorial(transcript: str, hf_token: str, model: str = DEFAULT_LLM) -> dict: """Call the chat model and return a normalized ``{title, intro, steps}`` dict.""" if not hf_token: raise ValueError("An HF token is required for the tutorial LLM (billed to your key).") truncated = transcript[:MAX_TRANSCRIPT_CHARS] if len(transcript) > MAX_TRANSCRIPT_CHARS: truncated += "\n[... transcript truncated for length ...]" prompt = _INSTRUCTIONS.replace("{transcript}", truncated) client = InferenceClient(token=hf_token) try: resp = client.chat.completions.create( model=model, messages=[ {"role": "system", "content": _SYSTEM}, {"role": "user", "content": prompt}, ], temperature=0.3, max_tokens=4000, ) except Exception as exc: raise RuntimeError(f"Tutorial LLM call failed ({model}): {exc}") from exc content = resp.choices[0].message.content or "" try: data = _extract_json(content) except Exception as exc: raise RuntimeError( f"Could not parse JSON from the LLM. First 400 chars:\n{content[:400]}" ) from exc return _normalize(data)