nyaya-agent / tools /ingest.py
SeriousSam07's picture
Upgrade: Nyaya Agent 2.0 - Judicial Lockdown, Aura Themes, and Export Suite
f26ea36
Raw
History Blame Contribute Delete
1.96 kB
import os
import chromadb
from chromadb.utils import embedding_functions
from pypdf import PdfReader
from dotenv import load_dotenv
load_dotenv(override=True)
# Configuration
CHROMA_PATH = "data/chroma_db"
COLLECTION_NAME = "nyaya_legal_docs"
# Initialize ChromaDB
client = chromadb.PersistentClient(path=CHROMA_PATH)
# Use Gemini embeddings via a custom wrapper or OpenAI-compatible one if available
# For simplicity in this workshop, we'll use a default sentence-transformer if available,
# or a simple placeholder.
# Better yet: Use Chroma's built-in default for now.
embedding_func = embedding_functions.DefaultEmbeddingFunction()
collection = client.get_or_create_collection(
name=COLLECTION_NAME,
embedding_function=embedding_func
)
def ingest_pdf(file_path):
"""Reads a PDF, chunks it, and adds it to the vector database."""
print(f"Ingesting: {file_path}")
reader = PdfReader(file_path)
doc_id = os.path.basename(file_path)
chunks = []
metadata = []
ids = []
for i, page in enumerate(reader.pages):
text = page.extract_text()
if text.strip():
# Basic chunking by page for now
chunks.append(text)
metadata.append({"source": doc_id, "page": i + 1})
ids.append(f"{doc_id}_page_{i+1}")
if chunks:
collection.add(
documents=chunks,
metadatas=metadata,
ids=ids
)
print(f"Successfully added {len(chunks)} pages from {doc_id}")
def search_docs(query, n_results=3):
"""Searches the collection for relevant snippets."""
results = collection.query(
query_texts=[query],
n_results=n_results
)
return results
if __name__ == "__main__":
# Test with any PDFs in the data folder
data_dir = "data"
for file in os.listdir(data_dir):
if file.endswith(".pdf"):
ingest_pdf(os.path.join(data_dir, file))