7th_handle / src /colab_uploader.py
Adoption's picture
feat: implement sermon message processing and RAG pipeline with Streamlit dashboard and data ingestion tools
0004cda
Raw
History Blame Contribute Delete
7 kB
# ==============================================================================
# Google Colab High-Speed Pinecone Sermon Ingestion Script (bge-base-en-v1.5)
# ==============================================================================
# This script is designed to run in a Google Colab notebook with a GPU (T4 or better).
# It will generate embeddings at over 500 chunks/sec, uploading your entire
# sermon database (102,127 chunks) in under 5 minutes!
#
# INSTRUCTIONS:
# 1. Open Google Colab: https://colab.research.google.com/
# 2. Go to: Runtime -> Change runtime type -> Select T4 GPU (or any available GPU).
# 3. Copy and paste this script into a code cell in Colab.
# 4. Upload your 'sermon_chunks.pkl' file to the root directory in Colab using
# the file explorer sidebar.
# 5. Run the code cell!
# ==============================================================================
import os
import sys
import json
import time
import pickle
from typing import List
# Install required dependencies inside the Colab environment
print("Installing high-speed dependencies...")
!pip install -q "numpy<2.0.0" langchain langchain-community langchain-pinecone sentence-transformers torch pinecone-client
import torch
from langchain_core.documents import Document
from langchain_community.embeddings import HuggingFaceEmbeddings
from langchain_pinecone import PineconeVectorStore
from pinecone import Pinecone
# --- CONFIGURATION ---
INDEX_NAME = "branham-index"
BATCH_SIZE = 128 # Double the batch size to maximize GPU memory throughput!
CHECKPOINT_FILE = "colab_upload_checkpoint.json"
def main():
# 1. Verify GPU availability
device = "cuda" if torch.cuda.is_available() else "cpu"
print(f"\n[DEVICE STATUS] PyTorch is running on: {device.upper()}")
if device == "cpu":
print("WARNING: GPU is not active! In Colab, go to 'Runtime' -> 'Change runtime type' and select GPU for 100x speedup.")
else:
print("SUCCESS: T4 GPU is active and ready to accelerate embeddings!")
# 2. Verify sermon_chunks.pkl file
chunks_file = "sermon_chunks.pkl"
if not os.path.exists(chunks_file):
print(f"\n[ERROR] '{chunks_file}' was not found in the Colab directory!")
print("Please drag and drop 'sermon_chunks.pkl' from your laptop into the Colab file explorer on the left, then run this cell again.")
return
print(f"\nLoading chunks from '{chunks_file}'...")
with open(chunks_file, "rb") as f:
chunks = pickle.load(f)
total_chunks = len(chunks)
print(f"Successfully loaded {total_chunks} sermon chunks.")
# 3. Prompt for Pinecone API Key
pinecone_key = input("\nEnter your PINECONE_API_KEY: ").strip()
if not pinecone_key:
print("[ERROR] Pinecone API Key is required.")
return
os.environ["PINECONE_API_KEY"] = pinecone_key
# 4. Connect to Pinecone and clear index (or resume)
pc = Pinecone(api_key=pinecone_key)
print(f"\nConnecting to Pinecone index '{INDEX_NAME}'...")
idx = pc.Index(INDEX_NAME)
# Check for resume checkpoint
start_idx = 0
if os.path.exists(CHECKPOINT_FILE):
try:
with open(CHECKPOINT_FILE, "r") as f:
checkpoint = json.load(f)
start_idx = checkpoint.get("last_uploaded_index", 0)
print(f"[RESUME] Found checkpoint! Resuming from chunk index {start_idx}...")
except Exception as e:
print(f"[INFO] Failed to read checkpoint, starting from scratch: {e}")
if start_idx == 0:
stats = idx.describe_index_stats()
total_vectors = stats.get("total_vector_count", 0)
print(f"Current total vector count: {total_vectors}")
clear_choice = input("Do you want to clear the index before uploading? (y/n): ").strip().lower()
if clear_choice == "y":
print(f"Deleting all vectors in '{INDEX_NAME}' to start fresh...")
try:
idx.delete(delete_all=True)
print("Index cleared successfully!")
time.sleep(5)
except Exception as e:
print(f"Index deletion skipped/failed: {e}")
# 5. Initialize Hugging Face BGE Model on CUDA GPU
print("\nInitializing Hugging Face BGE model (BAAI/bge-base-en-v1.5) on GPU...")
model_start = time.perf_counter()
embeddings = HuggingFaceEmbeddings(
model_name="BAAI/bge-base-en-v1.5",
model_kwargs={'device': device},
encode_kwargs={'normalize_embeddings': True}
)
vector_store = PineconeVectorStore(
index_name=INDEX_NAME,
embedding=embeddings,
)
print(f"Model and Vector Store initialized in {time.perf_counter() - model_start:.1f}s.")
# 6. High-Speed Upload
print(f"\nStarting batch upload of {total_chunks} chunks to Pinecone...")
print(f"Batch Size: {BATCH_SIZE} | GPU acceleration enabled: {device.upper()}\n")
start_time = time.perf_counter()
total_batches = (total_chunks + BATCH_SIZE - 1) // BATCH_SIZE
for i in range(start_idx, total_chunks, BATCH_SIZE):
batch = chunks[i : i + BATCH_SIZE]
batch_num = (i // BATCH_SIZE) + 1
# Robust upload loop
retries = 0
while retries <= 5:
try:
vector_store.add_documents(batch)
# Progress logging
elapsed = time.perf_counter() - start_time
pct = (min(i + BATCH_SIZE, total_chunks) / total_chunks) * 100
rate = (i + len(batch) - start_idx) / elapsed if elapsed > 0 else 0
eta = (total_chunks - (i + len(batch))) / rate if rate > 0 else 0
print(
f"[{pct:6.2f}%] Batch {batch_num}/{total_batches} uploaded successfully. "
f"({min(i + BATCH_SIZE, total_chunks)}/{total_chunks}) | "
f"Speed: {rate:.1f} chunks/sec | ETA: {eta/60:.1f} min"
)
# Write checkpoint
with open(CHECKPOINT_FILE, "w") as f:
json.dump({"last_uploaded_index": i + len(batch)}, f)
break
except Exception as e:
retries += 1
if retries > 5:
print(f"\n[FATAL ERROR] Batch {batch_num} failed completely. Error: {e}")
return
sleep_time = 2 ** retries
print(f"\n[WARNING] Error on Batch {batch_num} (Attempt {retries}/5): {e}. Retrying in {sleep_time}s...")
time.sleep(sleep_time)
print("\n==============================================================================")
print("🎉 UPLOAD COMPLETED SUCCESSFULLY! 🎉")
print(f"Uploaded {total_chunks} chunks to Pinecone in {time.perf_counter() - start_time:.1f}s.")
print("==============================================================================")
if __name__ == "__main__":
main()