organic-chatbot / ingest_documents.py
Maia Pelletier
add links for PDFs
588cdee
Raw
History Blame Contribute Delete
3.39 kB
"""
Standalone script to ingest PDFs and URLs into the vector store.
Run this script periodically to update your document knowledge base.
"""
import os
from pathlib import Path
from ingestion import DocumentIngestion
# Configuration
PDF_FOLDER = "data/pdfs" # Folder containing PDF files
URLS = [
# Add your URLs here, one per line
"https://inspection.canada.ca/en/food-labels/organic-products/operating-manual",
"https://inspection.canada.ca/en/food-labels/organic-products/import-requirements",
"https://inspection.canada.ca/en/food-labels/labelling/industry/organic-claims",
"https://cog.ca/faqs/"
]
# Optional: map PDF filenames to their publicly hosted URLs so references are hyperlinked.
# Keys are bare filenames (no path), values are the public URL for that PDF.
PDF_URLS = {
"Organic production systems - General principles and management standards.pdf": "https://publications.gc.ca/collections/collection_2026/ongc-cgsb/P29-32-310-2026-eng.pdf",
# "another-doc.pdf": "https://example.com/another-doc.pdf",
}
def main():
"""Main ingestion function."""
print("=" * 60)
print("Document Ingestion Script")
print("=" * 60)
# Initialize ingestion system
print("\nInitializing document ingestion system...")
ingestion = DocumentIngestion(embedding_model="all-mpnet-base-v2")
# Collect PDF files
pdf_paths = []
if os.path.exists(PDF_FOLDER):
pdf_files = list(Path(PDF_FOLDER).glob("*.pdf"))
pdf_paths = [str(f) for f in pdf_files]
print(f"\nFound {len(pdf_paths)} PDF file(s) in {PDF_FOLDER}:")
for pdf in pdf_paths:
print(f" - {os.path.basename(pdf)}")
else:
print(f"\nPDF folder '{PDF_FOLDER}' not found. Creating it...")
os.makedirs(PDF_FOLDER, exist_ok=True)
print(f"Please add PDF files to {PDF_FOLDER} and run again.")
# Filter out empty URLs
urls = [url.strip() for url in URLS if url.strip()]
if urls:
print(f"\nFound {len(urls)} URL(s) to process:")
for url in urls:
print(f" - {url}")
else:
print("\nNo URLs configured. Add URLs to the URLS list in this script.")
if not pdf_paths and not urls:
print("\n[ERROR] No documents to process. Please add PDFs or URLs.")
return
# Process documents
print("\n" + "=" * 60)
print("Processing documents...")
print("=" * 60)
try:
documents = ingestion.process_documents(pdf_paths=pdf_paths, urls=urls, pdf_urls=PDF_URLS)
print(f"\n[SUCCESS] Successfully processed {len(documents)} document chunks")
# Build vector store
print("\nBuilding vector store...")
ingestion.build_vector_store()
# Save vector store
print("\nSaving vector store...")
ingestion.save("data/vector_store")
print("\n" + "=" * 60)
print("[SUCCESS] Ingestion complete!")
print("=" * 60)
print(f"\nTotal document chunks: {len(documents)}")
print(f"Vector store saved to: data/vector_store")
print("\nYou can now run 'py app.py' to start the chatbot.")
except Exception as e:
print(f"\n[ERROR] Error during ingestion: {str(e)}")
import traceback
traceback.print_exc()
return
if __name__ == "__main__":
main()