import os from langchain_community.document_loaders import PyPDFLoader from langchain.text_splitter import RecursiveCharacterTextSplitter from langchain_community.embeddings import HuggingFaceEmbeddings from langchain_community.vectorstores import FAISS import pickle # Added for saving embeddings object # --- Configuration --- # Calculate absolute path to the PDF from the script's location SCRIPT_DIR = os.path.dirname(__file__) PROJECT_ROOT = os.path.abspath(os.path.join(SCRIPT_DIR, '..')) PDF_PATH = os.path.join(PROJECT_ROOT, "School Eligibility Rules.pdf") # PDF_PATH = "../School Eligibility Rules.pdf" # Original relative path INDEX_SAVE_PATH = os.path.join(PROJECT_ROOT, "faiss_index") # Save index in the root directory EMBEDDING_MODEL_NAME = "all-MiniLM-L6-v2" # A good, lightweight sentence transformer print("Starting vector store creation...") CHUNK_SIZE = 1000 # Characters per chunk CHUNK_OVERLAP = 150 # Overlap between chunks # --- 1. Load PDF --- print(f"Loading PDF from: {PDF_PATH}") if not os.path.exists(PDF_PATH): print(f"Error: PDF file not found at {PDF_PATH}") exit() loader = PyPDFLoader(PDF_PATH) documents = loader.load() print(f"Loaded {len(documents)} pages from PDF.") # --- 2. Split Text --- print(f"Splitting text into chunks (size={CHUNK_SIZE}, overlap={CHUNK_OVERLAP})...") text_splitter = RecursiveCharacterTextSplitter( chunk_size=CHUNK_SIZE, chunk_overlap=CHUNK_OVERLAP, length_function=len ) docs_split = text_splitter.split_documents(documents) print(f"Split into {len(docs_split)} text chunks.") if not docs_split: print("Error: No text chunks were generated. Check PDF content and splitting parameters.") exit() # --- 3. Create Embeddings --- print(f"Creating embeddings using model: {EMBEDDING_MODEL_NAME}...") # Specify device explicitly if needed (e.g., 'cuda' for GPU, 'cpu' for CPU) # Set cache_folder to avoid re-downloading if possible model_kwargs = {'device': 'cpu'} # Use CPU explicitly encode_kwargs = {'normalize_embeddings': False} # Normalization handled by FAISS if needed embeddings = HuggingFaceEmbeddings( model_name=EMBEDDING_MODEL_NAME, model_kwargs=model_kwargs, encode_kwargs=encode_kwargs, # cache_folder='./model_cache' # Optional: specify a cache directory ) # --- 4. Create FAISS Vector Store --- print("Creating FAISS vector store...") # FAISS.from_documents might be memory intensive for very large PDFs. # Consider processing in batches if needed. try: vectorstore = FAISS.from_documents(docs_split, embeddings) print("FAISS vector store created successfully.") except Exception as e: print(f"Error creating FAISS vector store: {e}") exit() # --- 5. Save Index --- print(f"Saving FAISS index to: {INDEX_SAVE_PATH}...") vectorstore.save_local(INDEX_SAVE_PATH) # Optional: Save the embeddings object itself if needed separately, though often not required # with open(f"{INDEX_SAVE_PATH}_embeddings.pkl", "wb") as f: # pickle.dump(embeddings, f) print("Vector store creation complete!") print(f"Index saved at: {INDEX_SAVE_PATH}")