Spaces:
Configuration error
Configuration error
| import os | |
| import glob | |
| from typing import List, Dict, Any, Optional | |
| import pypdf | |
| from app.config import DATA_DIR, PRECEDENCE_LEVELS | |
| from app.core.security import UserContext | |
| class IndexedDocument: | |
| pass | |
| class DocumentIndexer: | |
| def __init__(self, data_dir: str = str(DATA_DIR)): | |
| self.data_dir = data_dir | |
| self.documents: List[Dict[str, Any]] = [] | |
| self.load_and_index_documents() | |
| def load_and_index_documents(self): | |
| """Loads and indexes all PDF documents from the data directory with authority metadata.""" | |
| pdf_files = sorted(glob.glob(os.path.join(self.data_dir, "*.pdf"))) | |
| self.documents = [] | |
| for pdf_path in pdf_files: | |
| filename = os.path.basename(pdf_path) | |
| try: | |
| reader = pypdf.PdfReader(pdf_path) | |
| full_text = "\n".join([page.extract_text() or "" for page in reader.pages]) | |
| doc_type, status, level, account_id = self._classify_document(filename, full_text) | |
| doc_entry = { | |
| "filename": filename, | |
| "filepath": pdf_path, | |
| "title": filename.replace(".pdf", "").replace("_", " "), | |
| "content": full_text, | |
| "doc_type": doc_type, | |
| "status": status, | |
| "precedence_level": level, | |
| "account_id": account_id, | |
| "pages": len(reader.pages) | |
| } | |
| self.documents.append(doc_entry) | |
| except Exception as e: | |
| print(f"Error loading document {filename}: {e}") | |
| def _classify_document(self, filename: str, content: str): | |
| """Classifies document authority, status, precedence level, and account mapping.""" | |
| fn = filename.lower() | |
| if "v2_deprecated" in fn or "deprecated" in content.lower() and "do not use" in content.lower(): | |
| return "DEPRECATED_POLICY", "DEPRECATED", PRECEDENCE_LEVELS["DEPRECATED_POLICY"], None | |
| if "northstar" in fn: | |
| return "CUSTOMER_AGREEMENT", "CURRENT", PRECEDENCE_LEVELS["CUSTOMER_AGREEMENT"], "ACCT-001" | |
| elif "lumenworks" in fn: | |
| return "CUSTOMER_AGREEMENT", "CURRENT", PRECEDENCE_LEVELS["CUSTOMER_AGREEMENT"], "ACCT-002" | |
| elif "v3_current" in fn or "support policy v3" in content.lower(): | |
| return "CURRENT_SUPPORT_POLICY", "CURRENT", PRECEDENCE_LEVELS["CURRENT_SUPPORT_POLICY"], None | |
| elif "cancellation" in fn or "sop" in fn: | |
| return "CURRENT_SOP", "CURRENT", PRECEDENCE_LEVELS["CURRENT_SOP"], None | |
| elif "product_operations" in fn or "known_issues" in fn: | |
| return "PRODUCT_OPS_GUIDE", "CURRENT", PRECEDENCE_LEVELS["PRODUCT_OPS_GUIDE"], None | |
| else: | |
| return "GENERAL_DOC", "CURRENT", 1, None | |
| def search_documents( | |
| self, | |
| query: str, | |
| user_context: UserContext, | |
| include_deprecated: bool = False, | |
| top_k: int = 5 | |
| ) -> List[Dict[str, Any]]: | |
| """ | |
| Searches documents with keyword matching & precedence ranking. | |
| Strictly enforces access control (hides customer agreements of other accounts). | |
| Filters out DEPRECATED documents unless explicitly requested. | |
| """ | |
| query_terms = [t.lower() for t in query.split() if len(t) > 2] | |
| results = [] | |
| for doc in self.documents: | |
| # Access Control Filter | |
| if not user_context.can_access_document(doc["filename"], doc["account_id"]): | |
| continue | |
| # Deprecated Filter | |
| if doc["status"] == "DEPRECATED" and not include_deprecated: | |
| continue | |
| # Relevance Scoring | |
| content_lower = doc["content"].lower() | |
| title_lower = doc["title"].lower() | |
| score = 0 | |
| for term in query_terms: | |
| if term in title_lower: | |
| score += 10 | |
| score += content_lower.count(term) | |
| if score > 0 or not query_terms: | |
| results.append({ | |
| "doc": doc, | |
| "relevance_score": score, | |
| "precedence_level": doc["precedence_level"], | |
| "status": doc["status"], | |
| "account_id": doc["account_id"] | |
| }) | |
| # Sort primarily by precedence_level DESC (Higher authority first), then relevance_score DESC | |
| results.sort(key=lambda x: (x["precedence_level"], x["relevance_score"]), reverse=True) | |
| formatted_results = [] | |
| for r in results[:top_k]: | |
| doc = r["doc"] | |
| snippet = self._extract_snippet(doc["content"], query_terms) | |
| formatted_results.append({ | |
| "filename": doc["filename"], | |
| "title": doc["title"], | |
| "doc_type": doc["doc_type"], | |
| "precedence_level": doc["precedence_level"], | |
| "status": doc["status"], | |
| "account_id": doc["account_id"], | |
| "content_snippet": snippet, | |
| "full_content": doc["content"], | |
| "relevance_score": r["relevance_score"] | |
| }) | |
| return formatted_results | |
| def _extract_snippet(self, content: str, terms: List[str], max_len: int = 400) -> str: | |
| if not terms: | |
| return content[:max_len] + ("..." if len(content) > max_len else "") | |
| content_lower = content.lower() | |
| best_pos = 0 | |
| for term in terms: | |
| pos = content_lower.find(term) | |
| if pos != -1: | |
| best_pos = pos | |
| break | |
| start = max(0, best_pos - 50) | |
| end = min(len(content), start + max_len) | |
| return ( "..." if start > 0 else "" ) + content[start:end] + ( "..." if end < len(content) else "" ) | |
| def get_all_accessible_documents(self, user_context: UserContext) -> List[Dict[str, Any]]: | |
| """Returns list of all documents accessible to the given user context.""" | |
| return [ | |
| { | |
| "filename": d["filename"], | |
| "title": d["title"], | |
| "doc_type": d["doc_type"], | |
| "status": d["status"], | |
| "precedence_level": d["precedence_level"], | |
| "account_id": d["account_id"] | |
| } | |
| for d in self.documents | |
| if user_context.can_access_document(d["filename"], d["account_id"]) | |
| ] | |