Spaces:
Configuration error
Configuration error
File size: 7,002 Bytes
0004cda | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 | # ==============================================================================
# Google Colab High-Speed Pinecone Sermon Ingestion Script (bge-base-en-v1.5)
# ==============================================================================
# This script is designed to run in a Google Colab notebook with a GPU (T4 or better).
# It will generate embeddings at over 500 chunks/sec, uploading your entire
# sermon database (102,127 chunks) in under 5 minutes!
#
# INSTRUCTIONS:
# 1. Open Google Colab: https://colab.research.google.com/
# 2. Go to: Runtime -> Change runtime type -> Select T4 GPU (or any available GPU).
# 3. Copy and paste this script into a code cell in Colab.
# 4. Upload your 'sermon_chunks.pkl' file to the root directory in Colab using
# the file explorer sidebar.
# 5. Run the code cell!
# ==============================================================================
import os
import sys
import json
import time
import pickle
from typing import List
# Install required dependencies inside the Colab environment
print("Installing high-speed dependencies...")
!pip install -q "numpy<2.0.0" langchain langchain-community langchain-pinecone sentence-transformers torch pinecone-client
import torch
from langchain_core.documents import Document
from langchain_community.embeddings import HuggingFaceEmbeddings
from langchain_pinecone import PineconeVectorStore
from pinecone import Pinecone
# --- CONFIGURATION ---
INDEX_NAME = "branham-index"
BATCH_SIZE = 128 # Double the batch size to maximize GPU memory throughput!
CHECKPOINT_FILE = "colab_upload_checkpoint.json"
def main():
# 1. Verify GPU availability
device = "cuda" if torch.cuda.is_available() else "cpu"
print(f"\n[DEVICE STATUS] PyTorch is running on: {device.upper()}")
if device == "cpu":
print("WARNING: GPU is not active! In Colab, go to 'Runtime' -> 'Change runtime type' and select GPU for 100x speedup.")
else:
print("SUCCESS: T4 GPU is active and ready to accelerate embeddings!")
# 2. Verify sermon_chunks.pkl file
chunks_file = "sermon_chunks.pkl"
if not os.path.exists(chunks_file):
print(f"\n[ERROR] '{chunks_file}' was not found in the Colab directory!")
print("Please drag and drop 'sermon_chunks.pkl' from your laptop into the Colab file explorer on the left, then run this cell again.")
return
print(f"\nLoading chunks from '{chunks_file}'...")
with open(chunks_file, "rb") as f:
chunks = pickle.load(f)
total_chunks = len(chunks)
print(f"Successfully loaded {total_chunks} sermon chunks.")
# 3. Prompt for Pinecone API Key
pinecone_key = input("\nEnter your PINECONE_API_KEY: ").strip()
if not pinecone_key:
print("[ERROR] Pinecone API Key is required.")
return
os.environ["PINECONE_API_KEY"] = pinecone_key
# 4. Connect to Pinecone and clear index (or resume)
pc = Pinecone(api_key=pinecone_key)
print(f"\nConnecting to Pinecone index '{INDEX_NAME}'...")
idx = pc.Index(INDEX_NAME)
# Check for resume checkpoint
start_idx = 0
if os.path.exists(CHECKPOINT_FILE):
try:
with open(CHECKPOINT_FILE, "r") as f:
checkpoint = json.load(f)
start_idx = checkpoint.get("last_uploaded_index", 0)
print(f"[RESUME] Found checkpoint! Resuming from chunk index {start_idx}...")
except Exception as e:
print(f"[INFO] Failed to read checkpoint, starting from scratch: {e}")
if start_idx == 0:
stats = idx.describe_index_stats()
total_vectors = stats.get("total_vector_count", 0)
print(f"Current total vector count: {total_vectors}")
clear_choice = input("Do you want to clear the index before uploading? (y/n): ").strip().lower()
if clear_choice == "y":
print(f"Deleting all vectors in '{INDEX_NAME}' to start fresh...")
try:
idx.delete(delete_all=True)
print("Index cleared successfully!")
time.sleep(5)
except Exception as e:
print(f"Index deletion skipped/failed: {e}")
# 5. Initialize Hugging Face BGE Model on CUDA GPU
print("\nInitializing Hugging Face BGE model (BAAI/bge-base-en-v1.5) on GPU...")
model_start = time.perf_counter()
embeddings = HuggingFaceEmbeddings(
model_name="BAAI/bge-base-en-v1.5",
model_kwargs={'device': device},
encode_kwargs={'normalize_embeddings': True}
)
vector_store = PineconeVectorStore(
index_name=INDEX_NAME,
embedding=embeddings,
)
print(f"Model and Vector Store initialized in {time.perf_counter() - model_start:.1f}s.")
# 6. High-Speed Upload
print(f"\nStarting batch upload of {total_chunks} chunks to Pinecone...")
print(f"Batch Size: {BATCH_SIZE} | GPU acceleration enabled: {device.upper()}\n")
start_time = time.perf_counter()
total_batches = (total_chunks + BATCH_SIZE - 1) // BATCH_SIZE
for i in range(start_idx, total_chunks, BATCH_SIZE):
batch = chunks[i : i + BATCH_SIZE]
batch_num = (i // BATCH_SIZE) + 1
# Robust upload loop
retries = 0
while retries <= 5:
try:
vector_store.add_documents(batch)
# Progress logging
elapsed = time.perf_counter() - start_time
pct = (min(i + BATCH_SIZE, total_chunks) / total_chunks) * 100
rate = (i + len(batch) - start_idx) / elapsed if elapsed > 0 else 0
eta = (total_chunks - (i + len(batch))) / rate if rate > 0 else 0
print(
f"[{pct:6.2f}%] Batch {batch_num}/{total_batches} uploaded successfully. "
f"({min(i + BATCH_SIZE, total_chunks)}/{total_chunks}) | "
f"Speed: {rate:.1f} chunks/sec | ETA: {eta/60:.1f} min"
)
# Write checkpoint
with open(CHECKPOINT_FILE, "w") as f:
json.dump({"last_uploaded_index": i + len(batch)}, f)
break
except Exception as e:
retries += 1
if retries > 5:
print(f"\n[FATAL ERROR] Batch {batch_num} failed completely. Error: {e}")
return
sleep_time = 2 ** retries
print(f"\n[WARNING] Error on Batch {batch_num} (Attempt {retries}/5): {e}. Retrying in {sleep_time}s...")
time.sleep(sleep_time)
print("\n==============================================================================")
print("🎉 UPLOAD COMPLETED SUCCESSFULLY! 🎉")
print(f"Uploaded {total_chunks} chunks to Pinecone in {time.perf_counter() - start_time:.1f}s.")
print("==============================================================================")
if __name__ == "__main__":
main()
|