import os import glob from typing import List, Dict, Any, Optional import pypdf from app.config import DATA_DIR, PRECEDENCE_LEVELS from app.core.security import UserContext class IndexedDocument: pass class DocumentIndexer: def __init__(self, data_dir: str = str(DATA_DIR)): self.data_dir = data_dir self.documents: List[Dict[str, Any]] = [] self.load_and_index_documents() def load_and_index_documents(self): """Loads and indexes all PDF documents from the data directory with authority metadata.""" pdf_files = sorted(glob.glob(os.path.join(self.data_dir, "*.pdf"))) self.documents = [] for pdf_path in pdf_files: filename = os.path.basename(pdf_path) try: reader = pypdf.PdfReader(pdf_path) full_text = "\n".join([page.extract_text() or "" for page in reader.pages]) doc_type, status, level, account_id = self._classify_document(filename, full_text) doc_entry = { "filename": filename, "filepath": pdf_path, "title": filename.replace(".pdf", "").replace("_", " "), "content": full_text, "doc_type": doc_type, "status": status, "precedence_level": level, "account_id": account_id, "pages": len(reader.pages) } self.documents.append(doc_entry) except Exception as e: print(f"Error loading document {filename}: {e}") def _classify_document(self, filename: str, content: str): """Classifies document authority, status, precedence level, and account mapping.""" fn = filename.lower() if "v2_deprecated" in fn or "deprecated" in content.lower() and "do not use" in content.lower(): return "DEPRECATED_POLICY", "DEPRECATED", PRECEDENCE_LEVELS["DEPRECATED_POLICY"], None if "northstar" in fn: return "CUSTOMER_AGREEMENT", "CURRENT", PRECEDENCE_LEVELS["CUSTOMER_AGREEMENT"], "ACCT-001" elif "lumenworks" in fn: return "CUSTOMER_AGREEMENT", "CURRENT", PRECEDENCE_LEVELS["CUSTOMER_AGREEMENT"], "ACCT-002" elif "v3_current" in fn or "support policy v3" in content.lower(): return "CURRENT_SUPPORT_POLICY", "CURRENT", PRECEDENCE_LEVELS["CURRENT_SUPPORT_POLICY"], None elif "cancellation" in fn or "sop" in fn: return "CURRENT_SOP", "CURRENT", PRECEDENCE_LEVELS["CURRENT_SOP"], None elif "product_operations" in fn or "known_issues" in fn: return "PRODUCT_OPS_GUIDE", "CURRENT", PRECEDENCE_LEVELS["PRODUCT_OPS_GUIDE"], None else: return "GENERAL_DOC", "CURRENT", 1, None def search_documents( self, query: str, user_context: UserContext, include_deprecated: bool = False, top_k: int = 5 ) -> List[Dict[str, Any]]: """ Searches documents with keyword matching & precedence ranking. Strictly enforces access control (hides customer agreements of other accounts). Filters out DEPRECATED documents unless explicitly requested. """ query_terms = [t.lower() for t in query.split() if len(t) > 2] results = [] for doc in self.documents: # Access Control Filter if not user_context.can_access_document(doc["filename"], doc["account_id"]): continue # Deprecated Filter if doc["status"] == "DEPRECATED" and not include_deprecated: continue # Relevance Scoring content_lower = doc["content"].lower() title_lower = doc["title"].lower() score = 0 for term in query_terms: if term in title_lower: score += 10 score += content_lower.count(term) if score > 0 or not query_terms: results.append({ "doc": doc, "relevance_score": score, "precedence_level": doc["precedence_level"], "status": doc["status"], "account_id": doc["account_id"] }) # Sort primarily by precedence_level DESC (Higher authority first), then relevance_score DESC results.sort(key=lambda x: (x["precedence_level"], x["relevance_score"]), reverse=True) formatted_results = [] for r in results[:top_k]: doc = r["doc"] snippet = self._extract_snippet(doc["content"], query_terms) formatted_results.append({ "filename": doc["filename"], "title": doc["title"], "doc_type": doc["doc_type"], "precedence_level": doc["precedence_level"], "status": doc["status"], "account_id": doc["account_id"], "content_snippet": snippet, "full_content": doc["content"], "relevance_score": r["relevance_score"] }) return formatted_results def _extract_snippet(self, content: str, terms: List[str], max_len: int = 400) -> str: if not terms: return content[:max_len] + ("..." if len(content) > max_len else "") content_lower = content.lower() best_pos = 0 for term in terms: pos = content_lower.find(term) if pos != -1: best_pos = pos break start = max(0, best_pos - 50) end = min(len(content), start + max_len) return ( "..." if start > 0 else "" ) + content[start:end] + ( "..." if end < len(content) else "" ) def get_all_accessible_documents(self, user_context: UserContext) -> List[Dict[str, Any]]: """Returns list of all documents accessible to the given user context.""" return [ { "filename": d["filename"], "title": d["title"], "doc_type": d["doc_type"], "status": d["status"], "precedence_level": d["precedence_level"], "account_id": d["account_id"] } for d in self.documents if user_context.can_access_document(d["filename"], d["account_id"]) ]