"""Stage 2-3 of the pipeline: rule-based pre-classifier and the GPT-4o-mini intent classifier.""" import asyncio import html import json import logging import os import re import time import uuid from datetime import datetime from typing import Any, Dict, List, Literal, Optional, Tuple from urllib.parse import quote import httpx from pydantic import BaseModel, ConfigDict, Field from src.config import get_settings, LIBBEE_VERSION from src.agentcore.models import ChatMessage from src.agentcore.constants import ( CURRENT_INFO_RE, GREETING_FOLLOWUP_RE, HISTORY_WINDOW, HOURS_RE, KU_CAMPUS_RE, LIBRARY_CUE_RE, MEDICAL_KEYWORDS, MEDICAL_SEARCH_RE, RESEARCH_CUE_RE, SOCIAL_RE, SUMMARY_RE, _CLASSIFIER_SYSTEM, ) from src.agentcore.utils import _get_llm logger = logging.getLogger(__name__) def _is_summary_request(question: str) -> bool: return bool(SUMMARY_RE.search(question or "")) def _is_greeting_menu_followup(question: str, history: List[ChatMessage]) -> bool: if not GREETING_FOLLOWUP_RE.match((question or "").strip()): return False for m in reversed(history): if m.role == "assistant": last = m.content.lower() return "i'm libbee" in last or "are you looking for one of these" in last return False def _looks_library_hours_question(question: str) -> bool: q = (question or "").strip().lower() return bool(HOURS_RE.search(q)) and any( kw in q for kw in ( "library", "campus", "habshan", "san", "hours", "open", "close", "schedule", "till", "until", "today", "tomorrow", "friday", "saturday", "sunday", "week", "ramadan", ) ) def _looks_campus_question(question: str) -> bool: return bool(KU_CAMPUS_RE.search(question or "")) def _looks_medical_search(question: str) -> bool: q = (question or "").lower() if not any(kw in q for kw in MEDICAL_KEYWORDS): return False return bool(MEDICAL_SEARCH_RE.search(q) or RESEARCH_CUE_RE.search(q) or SUMMARY_RE.search(q)) def _looks_research_question(question: str) -> bool: q = question or "" return bool(RESEARCH_CUE_RE.search(q) or SUMMARY_RE.search(q)) def _rule_based_classify(question: str) -> Dict[str, str]: q = (question or "").lower() if SOCIAL_RE.match((question or "").strip()): return { "intent": "social", "casual_answer": ( "Hello! I'm LibBee, the KU Library AI Assistant. " "I'm here to help you with articles, books, databases, and library services. " "What would you like to find today?" ), } if _looks_medical_search(question): return {"intent": "search_medical"} if _looks_research_question(question): return {"intent": "search_academic"} if LIBRARY_CUE_RE.search(q): return {"intent": "library_info"} if CURRENT_INFO_RE.search(q): return {"intent": "general_recent"} return {"intent": "general"} async def _llm_classify(question: str, history: List[ChatMessage], model: str) -> Dict[str, str]: settings = get_settings() if not settings.openai_api_key and not settings.anthropic_api_key: return _rule_based_classify(question) try: ctx_parts = [ f"{m.role}: {m.content[:100]}" for m in history[-HISTORY_WINDOW:] if m.role in ("user", "assistant") ] ctx = " | ".join(ctx_parts) if ctx_parts else "none" classify_model = "gpt" if settings.openai_api_key else model llm = _get_llm(classify_model, temperature=0, max_tokens=120) response = await llm.ainvoke([ {"role": "system", "content": _CLASSIFIER_SYSTEM}, {"role": "user", "content": f"Message: {question[:400]}\nContext: {ctx}\nJSON:"}, ]) raw = response.content.strip() if raw.startswith("```"): raw = raw.split("\n", 1)[1].rsplit("```", 1)[0].strip() result = json.loads(raw) intent = result.get("intent", "general") if intent not in {"social", "library_info", "search_academic", "search_medical", "general_recent", "general", "sensitive"}: intent = "general" return {"intent": intent, "casual_answer": result.get("casual_answer", "")} except Exception as e: logger.warning(f"LLM classifier error: {e} — falling back to rule-based") return _rule_based_classify(question)