Spaces:
Running on Zero
Running on Zero
| """ | |
| ui_helpers.py | |
| --------------------- | |
| Gradio UI support functions. | |
| _safe_content — defensive Gradio 6.x content extraction | |
| get_history_text — formats history list for the generation prompt | |
| build_retrieval_query — context-enriched query for follow-up questions | |
| format_metrics — builds the right-hand metrics panel markdown | |
| """ | |
| from config import CONFIDENCE_THRESHOLD, MAX_PROMPT_TOKENS | |
| def _safe_content(raw) -> str: | |
| """ | |
| Safely convert any Gradio message content value to a plain string. | |
| In Gradio 6.x, 'content' can be str, list of typed blocks | |
| (e.g. [{"type": "text", "text": "..."}]), or None. Calling | |
| .strip() directly on a list raises AttributeError — this | |
| function centralises the defensive conversion. | |
| """ | |
| if raw is None: | |
| return "" | |
| if isinstance(raw, str): | |
| return raw.strip() | |
| if isinstance(raw, list): | |
| parts = [] | |
| for block in raw: | |
| if isinstance(block, dict): | |
| parts.append(block.get("text", block.get("content", ""))) | |
| elif isinstance(block, str): | |
| parts.append(block) | |
| return " ".join(filter(None, parts)).strip() | |
| return str(raw).strip() | |
| def get_history_text(history: list, max_turns: int = 3) -> str: | |
| """ | |
| Convert Gradio's history list to a readable conversation string | |
| for inclusion in the generation prompt. | |
| max_turns=3 keeps enough context to resolve references without | |
| overflowing the prompt budget. UI status placeholders | |
| ("🔍 Searching...", etc.) are filtered out so they never leak | |
| into the model's view of the conversation. | |
| """ | |
| if not history: | |
| return "" | |
| UI_PREFIXES = ("🔍", "📚", "🌐", "❌", "⏳") | |
| lines = [] | |
| recent = history[-(max_turns * 2):] | |
| for msg in recent: | |
| if isinstance(msg, dict): | |
| role = "User" if msg.get("role") == "user" else "Assistant" | |
| content = _safe_content(msg.get("content")) | |
| elif isinstance(msg, (list, tuple)) and len(msg) == 2: | |
| for role, raw in [("User", msg[0]), ("Assistant", msg[1])]: | |
| content = _safe_content(raw) | |
| if content and not content.startswith(UI_PREFIXES): | |
| lines.append(f"{role}: {content}") | |
| continue | |
| else: | |
| continue | |
| if content and not content.startswith(UI_PREFIXES): | |
| lines.append(f"{role}: {content}") | |
| return "\n".join(lines) | |
| def build_retrieval_query(message: str, history: list) -> str: | |
| """ | |
| Enrich the FAISS/Tavily query with prior-turn context — but ONLY | |
| when the current message looks like a genuine follow-up question. | |
| Requires ALL three signals: short message (≤7 words), a reference | |
| word (pronoun/article), and a question starter that presupposes | |
| a subject. This avoids polluting retrieval for brand-new topics. | |
| """ | |
| if not history: | |
| return message | |
| words = message.lower().split() | |
| REFERENCE_WORDS = { | |
| "it", "they", "he", "she", "them", "those", "that", | |
| "this", "his", "her", "its", "their", "the", | |
| } | |
| QUESTION_STARTERS = { | |
| "who", "when", "where", "how", "why", | |
| "was", "did", "is", "are", | |
| } | |
| is_short = len(words) <= 7 | |
| has_reference = any(w in REFERENCE_WORDS for w in words) | |
| has_q_word = bool(words) and words[0] in QUESTION_STARTERS | |
| if not (is_short and has_reference and has_q_word): | |
| return message | |
| last_user_msg = "" | |
| for msg in reversed(history): | |
| if isinstance(msg, dict) and msg.get("role") == "user": | |
| last_user_msg = _safe_content(msg.get("content")) | |
| break | |
| elif isinstance(msg, (list, tuple)) and len(msg) >= 1: | |
| last_user_msg = _safe_content(msg[0]) | |
| break | |
| if last_user_msg and last_user_msg.lower() != message.lower(): | |
| return f"{last_user_msg} {message}" | |
| return message | |
| def format_metrics( | |
| source_type: str = "none", | |
| confidence: float = 0.0, | |
| sources_count: int = 0, | |
| prompt_tokens: int = 0, | |
| t_retrieve: float = 0.0, | |
| t_web: float = 0.0, | |
| t_generate: float = 0.0, | |
| t_total: float = 0.0, | |
| tokens_per_sec: float = 0.0, | |
| output_tokens: int = 0, | |
| generating: bool = False, | |
| elapsed: float = 0.0, | |
| token_so_far: int = 0, | |
| error: str = "", | |
| ) -> str: | |
| """ | |
| Return a markdown string for the right-hand metrics panel. | |
| Three display modes: empty (no query yet), generating (live view | |
| during streaming), done (full breakdown after completion). | |
| """ | |
| if error: | |
| return f"❌ **Error:** `{error}`" | |
| if source_type == "none" and not generating and t_total == 0.0: | |
| return "_Metrics will appear here during and after the response._" | |
| if generating: | |
| web_line = f"- Web search: **{t_web:.1f}s**\n" if t_web > 0 else "" | |
| return ( | |
| f"⚡ **Generating...**\n\n" | |
| f"- Tokens out: **{token_so_far}**\n" | |
| f"- Elapsed: **{elapsed:.1f}s**\n" | |
| f"- Retrieval: **{t_retrieve*1000:.0f} ms**\n" | |
| f"{web_line}" | |
| ) | |
| source_icon = "📚" if source_type == "faiss" else "🌐" | |
| source_name = "FAISS" if source_type == "faiss" else "Tavily" | |
| conf_status = "✓" if confidence >= CONFIDENCE_THRESHOLD else "✗ → web" | |
| lines = [ | |
| f"**{source_icon} {source_name}** — sim `{confidence:.3f}` {conf_status}", | |
| "", | |
| "**⏱ Timing**", | |
| f"- Retrieval: `{t_retrieve * 1000:.0f} ms`", | |
| ] | |
| if t_web > 0: | |
| lines.append(f"- Web search: `{t_web:.2f}s`") | |
| lines += [ | |
| f"- Generation: `{t_generate:.2f}s`", | |
| f"- **Total: `{t_total:.2f}s`**", | |
| "", | |
| "**🔢 Generation**", | |
| f"- Speed: **`{tokens_per_sec:.1f} tok/s`**", | |
| f"- Output tokens: `{output_tokens}`", | |
| f"- Prompt tokens: `{prompt_tokens}` / `{MAX_PROMPT_TOKENS}`", | |
| ] | |
| if sources_count > 0: | |
| lines.append(f"- Sources indexed: `{sources_count}`") | |
| return "\n".join(lines) | |