test / app /core /document_indexer.py
Anish
Deploy ParcelPilot AI with Git LFS
2567e7e
Raw
History Blame Contribute Delete
6.43 kB
import os
import glob
from typing import List, Dict, Any, Optional
import pypdf
from app.config import DATA_DIR, PRECEDENCE_LEVELS
from app.core.security import UserContext
class IndexedDocument:
pass
class DocumentIndexer:
def __init__(self, data_dir: str = str(DATA_DIR)):
self.data_dir = data_dir
self.documents: List[Dict[str, Any]] = []
self.load_and_index_documents()
def load_and_index_documents(self):
"""Loads and indexes all PDF documents from the data directory with authority metadata."""
pdf_files = sorted(glob.glob(os.path.join(self.data_dir, "*.pdf")))
self.documents = []
for pdf_path in pdf_files:
filename = os.path.basename(pdf_path)
try:
reader = pypdf.PdfReader(pdf_path)
full_text = "\n".join([page.extract_text() or "" for page in reader.pages])
doc_type, status, level, account_id = self._classify_document(filename, full_text)
doc_entry = {
"filename": filename,
"filepath": pdf_path,
"title": filename.replace(".pdf", "").replace("_", " "),
"content": full_text,
"doc_type": doc_type,
"status": status,
"precedence_level": level,
"account_id": account_id,
"pages": len(reader.pages)
}
self.documents.append(doc_entry)
except Exception as e:
print(f"Error loading document {filename}: {e}")
def _classify_document(self, filename: str, content: str):
"""Classifies document authority, status, precedence level, and account mapping."""
fn = filename.lower()
if "v2_deprecated" in fn or "deprecated" in content.lower() and "do not use" in content.lower():
return "DEPRECATED_POLICY", "DEPRECATED", PRECEDENCE_LEVELS["DEPRECATED_POLICY"], None
if "northstar" in fn:
return "CUSTOMER_AGREEMENT", "CURRENT", PRECEDENCE_LEVELS["CUSTOMER_AGREEMENT"], "ACCT-001"
elif "lumenworks" in fn:
return "CUSTOMER_AGREEMENT", "CURRENT", PRECEDENCE_LEVELS["CUSTOMER_AGREEMENT"], "ACCT-002"
elif "v3_current" in fn or "support policy v3" in content.lower():
return "CURRENT_SUPPORT_POLICY", "CURRENT", PRECEDENCE_LEVELS["CURRENT_SUPPORT_POLICY"], None
elif "cancellation" in fn or "sop" in fn:
return "CURRENT_SOP", "CURRENT", PRECEDENCE_LEVELS["CURRENT_SOP"], None
elif "product_operations" in fn or "known_issues" in fn:
return "PRODUCT_OPS_GUIDE", "CURRENT", PRECEDENCE_LEVELS["PRODUCT_OPS_GUIDE"], None
else:
return "GENERAL_DOC", "CURRENT", 1, None
def search_documents(
self,
query: str,
user_context: UserContext,
include_deprecated: bool = False,
top_k: int = 5
) -> List[Dict[str, Any]]:
"""
Searches documents with keyword matching & precedence ranking.
Strictly enforces access control (hides customer agreements of other accounts).
Filters out DEPRECATED documents unless explicitly requested.
"""
query_terms = [t.lower() for t in query.split() if len(t) > 2]
results = []
for doc in self.documents:
# Access Control Filter
if not user_context.can_access_document(doc["filename"], doc["account_id"]):
continue
# Deprecated Filter
if doc["status"] == "DEPRECATED" and not include_deprecated:
continue
# Relevance Scoring
content_lower = doc["content"].lower()
title_lower = doc["title"].lower()
score = 0
for term in query_terms:
if term in title_lower:
score += 10
score += content_lower.count(term)
if score > 0 or not query_terms:
results.append({
"doc": doc,
"relevance_score": score,
"precedence_level": doc["precedence_level"],
"status": doc["status"],
"account_id": doc["account_id"]
})
# Sort primarily by precedence_level DESC (Higher authority first), then relevance_score DESC
results.sort(key=lambda x: (x["precedence_level"], x["relevance_score"]), reverse=True)
formatted_results = []
for r in results[:top_k]:
doc = r["doc"]
snippet = self._extract_snippet(doc["content"], query_terms)
formatted_results.append({
"filename": doc["filename"],
"title": doc["title"],
"doc_type": doc["doc_type"],
"precedence_level": doc["precedence_level"],
"status": doc["status"],
"account_id": doc["account_id"],
"content_snippet": snippet,
"full_content": doc["content"],
"relevance_score": r["relevance_score"]
})
return formatted_results
def _extract_snippet(self, content: str, terms: List[str], max_len: int = 400) -> str:
if not terms:
return content[:max_len] + ("..." if len(content) > max_len else "")
content_lower = content.lower()
best_pos = 0
for term in terms:
pos = content_lower.find(term)
if pos != -1:
best_pos = pos
break
start = max(0, best_pos - 50)
end = min(len(content), start + max_len)
return ( "..." if start > 0 else "" ) + content[start:end] + ( "..." if end < len(content) else "" )
def get_all_accessible_documents(self, user_context: UserContext) -> List[Dict[str, Any]]:
"""Returns list of all documents accessible to the given user context."""
return [
{
"filename": d["filename"],
"title": d["title"],
"doc_type": d["doc_type"],
"status": d["status"],
"precedence_level": d["precedence_level"],
"account_id": d["account_id"]
}
for d in self.documents
if user_context.can_access_document(d["filename"], d["account_id"])
]