"""Lightweight, dependency-free summary + Q&A with an optional LLM upgrade. Base mode (no API key, no model): * Extractive summary — ranks transcript segments by domain-keyword density and bookmark coverage, returns the top sentences in chronological order. * Q&A — keyword retrieval over the timeline; answers with the best-matching transcript snippet and its timestamp ("grounded answer"). Optional mode: * If ``LLM_API_KEY`` is set, callers may plug in a real LLM. The hook is intentionally isolated so the base app never imports an LLM SDK. """ from __future__ import annotations from typing import Dict, List from src.config import Config, CONFIG from src.search import VideoSearch from src.utils import extract_keywords def extractive_summary( timeline_doc: Dict[str, object], config: Config = CONFIG, max_points: int = 6, ) -> Dict[str, object]: """Return a few key sentences ranked by keyword density + bookmark signal.""" entries: List[Dict[str, object]] = list(timeline_doc.get("timeline", [])) scored = [] for e in entries: text = str(e.get("transcript", "")) kw = extract_keywords(text, config.domain_terms) score = len(kw) + (2 if e.get("bookmark") else 0) if score > 0 and text: scored.append((score, e.get("time_sec", 0.0), e.get("time_label"), text, kw)) scored.sort(key=lambda x: x[0], reverse=True) top = scored[:max_points] top.sort(key=lambda x: x[1]) # back to chronological order points = [ {"time_label": t[2], "text": t[3], "keywords": t[4]} for t in top ] # Collect the distinct topics mentioned across the whole video. topics: List[str] = [] for e in entries: for k in extract_keywords(str(e.get("transcript", "")), config.domain_terms): if k not in topics: topics.append(k) return { "video_id": timeline_doc.get("video_id", ""), "mode": "extractive", "summary_points": points, "topics": topics, } def answer_question( question: str, timeline_doc: Dict[str, object], config: Config = CONFIG, top_k: int = 3, ) -> Dict[str, object]: """Grounded Q&A: retrieve the best timeline snippets for the question.""" engine = VideoSearch(timeline_doc, config) found = engine.search(question, top_k=top_k) results = found["results"] if results: best = results[0] answer = ( f"At {best['time_label']}: \"{best['text']}\"" if best.get("text") else f"See {best['time_label']}." ) else: answer = "No relevant moment was found in this video for that question." return { "question": question, "video_id": timeline_doc.get("video_id", ""), "mode": found["mode"], "answer": answer, "evidence": results, "llm_used": False, }