bps-eligibility / src /create_vector_store.py
Hara
RAG baseline - but dont use its not better than API
1681962
Raw
History Blame Contribute Delete
3.08 kB
import os
from langchain_community.document_loaders import PyPDFLoader
from langchain.text_splitter import RecursiveCharacterTextSplitter
from langchain_community.embeddings import HuggingFaceEmbeddings
from langchain_community.vectorstores import FAISS
import pickle # Added for saving embeddings object
# --- Configuration ---
# Calculate absolute path to the PDF from the script's location
SCRIPT_DIR = os.path.dirname(__file__)
PROJECT_ROOT = os.path.abspath(os.path.join(SCRIPT_DIR, '..'))
PDF_PATH = os.path.join(PROJECT_ROOT, "School Eligibility Rules.pdf")
# PDF_PATH = "../School Eligibility Rules.pdf" # Original relative path
INDEX_SAVE_PATH = os.path.join(PROJECT_ROOT, "faiss_index") # Save index in the root directory
EMBEDDING_MODEL_NAME = "all-MiniLM-L6-v2" # A good, lightweight sentence transformer
print("Starting vector store creation...")
CHUNK_SIZE = 1000 # Characters per chunk
CHUNK_OVERLAP = 150 # Overlap between chunks
# --- 1. Load PDF ---
print(f"Loading PDF from: {PDF_PATH}")
if not os.path.exists(PDF_PATH):
print(f"Error: PDF file not found at {PDF_PATH}")
exit()
loader = PyPDFLoader(PDF_PATH)
documents = loader.load()
print(f"Loaded {len(documents)} pages from PDF.")
# --- 2. Split Text ---
print(f"Splitting text into chunks (size={CHUNK_SIZE}, overlap={CHUNK_OVERLAP})...")
text_splitter = RecursiveCharacterTextSplitter(
chunk_size=CHUNK_SIZE,
chunk_overlap=CHUNK_OVERLAP,
length_function=len
)
docs_split = text_splitter.split_documents(documents)
print(f"Split into {len(docs_split)} text chunks.")
if not docs_split:
print("Error: No text chunks were generated. Check PDF content and splitting parameters.")
exit()
# --- 3. Create Embeddings ---
print(f"Creating embeddings using model: {EMBEDDING_MODEL_NAME}...")
# Specify device explicitly if needed (e.g., 'cuda' for GPU, 'cpu' for CPU)
# Set cache_folder to avoid re-downloading if possible
model_kwargs = {'device': 'cpu'} # Use CPU explicitly
encode_kwargs = {'normalize_embeddings': False} # Normalization handled by FAISS if needed
embeddings = HuggingFaceEmbeddings(
model_name=EMBEDDING_MODEL_NAME,
model_kwargs=model_kwargs,
encode_kwargs=encode_kwargs,
# cache_folder='./model_cache' # Optional: specify a cache directory
)
# --- 4. Create FAISS Vector Store ---
print("Creating FAISS vector store...")
# FAISS.from_documents might be memory intensive for very large PDFs.
# Consider processing in batches if needed.
try:
vectorstore = FAISS.from_documents(docs_split, embeddings)
print("FAISS vector store created successfully.")
except Exception as e:
print(f"Error creating FAISS vector store: {e}")
exit()
# --- 5. Save Index ---
print(f"Saving FAISS index to: {INDEX_SAVE_PATH}...")
vectorstore.save_local(INDEX_SAVE_PATH)
# Optional: Save the embeddings object itself if needed separately, though often not required
# with open(f"{INDEX_SAVE_PATH}_embeddings.pkl", "wb") as f:
# pickle.dump(embeddings, f)
print("Vector store creation complete!")
print(f"Index saved at: {INDEX_SAVE_PATH}")