Spaces:
Sleeping
Sleeping
| import gradio as gr | |
| import os | |
| import time | |
| import re | |
| import pandas as pd | |
| from dotenv import load_dotenv | |
| from pinecone import Pinecone, ServerlessSpec | |
| # --- Langchain components for document processing and embedding --- | |
| from langchain_openai import OpenAIEmbeddings | |
| from langchain_core.documents import Document | |
| from langchain_community.document_loaders import PyPDFLoader | |
| from langchain_text_splitters import RecursiveCharacterTextSplitter | |
| # Load environment variables from a .env file | |
| load_dotenv() | |
| # --- Backend Functions --- | |
| import traceback # Assicurati di aggiungere questo import all'inizio del file | |
| def get_stored_files(): | |
| """Retrieve a list of files currently stored in the Pinecone index with enhanced debugging.""" | |
| try: | |
| api_key = os.environ.get("PINECONE_API_KEY") | |
| index_name = os.environ.get("PINECONE_INDEX_NAME") | |
| # --- LOG DI DEBUG --- | |
| print("--- DEBUG: Esecuzione di get_stored_files ---") | |
| if not api_key: | |
| print("--- DEBUG ERROR: La variabile PINECONE_API_KEY non è impostata!") | |
| return [] | |
| if not index_name: | |
| print("--- DEBUG ERROR: La variabile PINECONE_INDEX_NAME non è impostata!") | |
| return [] | |
| print(f"--- DEBUG: Tento la connessione all'indice Pinecone: '{index_name}' ---") | |
| # --- FINE LOG DI DEBUG --- | |
| pc = Pinecone(api_key=api_key) | |
| # Controlla se l'indice esiste veramente | |
| existing_indexes = [index_info["name"] for index_info in pc.list_indexes()] | |
| print(f"--- DEBUG: Indici trovati nell'account: {existing_indexes} ---") | |
| if index_name not in existing_indexes: | |
| print(f"--- DEBUG WARNING: L'indice '{index_name}' non esiste. Restituisco una lista vuota. ---") | |
| return [] | |
| index = pc.Index(index_name) | |
| # Controlla le statistiche dell'indice per vedere se contiene vettori | |
| stats = index.describe_index_stats() | |
| print(f"--- DEBUG: Statistiche dell'indice '{index_name}': {stats} ---") | |
| if stats.get('total_vector_count', 0) == 0: | |
| print(f"--- DEBUG INFO: L'indice '{index_name}' è vuoto. Restituisco una lista vuota. ---") | |
| return [] | |
| # Se l'indice non è vuoto, procedi con la query | |
| results = index.query(vector=[0.0] * 3072, top_k=10000, include_metadata=True) | |
| unique_files = set() | |
| if results.matches: | |
| for match in results.matches: | |
| if 'metadata' in match and 'source' in match.metadata: | |
| unique_files.add(match.metadata['source']) | |
| print(f"--- DEBUG: File unici trovati: {list(unique_files)} ---") | |
| return sorted(list(unique_files)) | |
| except Exception as e: | |
| print(f"--- DEBUG EXCEPTION: Errore critico durante il recupero dei file! ---") | |
| # Stampa l'errore completo per un'analisi dettagliata | |
| traceback.print_exc() | |
| return [] | |
| def delete_file_from_vectorstore(filename): | |
| """Deletes all vectors associated with a specific filename from Pinecone using the pinecone library.""" | |
| if not filename: | |
| return "No file selected for deletion.", get_files_df() | |
| try: | |
| pc = Pinecone(api_key=os.environ.get("PINECONE_API_KEY")) | |
| index_name = os.environ.get("PINECONE_INDEX_NAME") | |
| index = pc.Index(index_name) | |
| # Use metadata filtering to delete all vectors associated with the file. | |
| index.delete(filter={"source": {"$eq": filename}}) | |
| return f"Successfully deleted {filename}.", get_files_df() | |
| except Exception as e: | |
| return f"Error while deleting the file: {str(e)}", get_files_df() | |
| def embedder(uploaded_file_path): | |
| """ | |
| Handles the embedding of the uploaded PDF file using langchain for processing | |
| and the pinecone library for vector store operations. | |
| """ | |
| if uploaded_file_path is None: | |
| return "No file uploaded. Please upload a PDF.", get_files_df() | |
| try: | |
| original_filename = os.path.basename(uploaded_file_path) | |
| pc = Pinecone(api_key=os.environ.get("PINECONE_API_KEY")) | |
| index_name = os.environ.get("PINECONE_INDEX_NAME") | |
| embedding_dimension = 3072 # As specified for text-embedding-3-large | |
| # Create the index if it doesn't exist | |
| if index_name not in [index_info["name"] for index_info in pc.list_indexes()]: | |
| pc.create_index( | |
| name=index_name, | |
| dimension=embedding_dimension, | |
| metric="cosine", | |
| spec=ServerlessSpec(cloud="aws", region="us-east-1"), | |
| ) | |
| while not pc.describe_index(index_name).status["ready"]: | |
| time.sleep(1) | |
| index = pc.Index(index_name) | |
| # 1. Load and Split Document | |
| loader = PyPDFLoader(uploaded_file_path) | |
| raw_documents = loader.load() | |
| text_splitter = RecursiveCharacterTextSplitter( | |
| chunk_size=800, | |
| chunk_overlap=400, | |
| length_function=len, | |
| ) | |
| documents = text_splitter.split_documents(raw_documents) | |
| # 2. Create Embeddings | |
| embeddings_model = OpenAIEmbeddings(model="text-embedding-3-large", api_key=os.environ.get("OPENAI_API_KEY")) | |
| texts_to_embed = [doc.page_content for doc in documents] | |
| embeddings = embeddings_model.embed_documents(texts_to_embed) | |
| # 3. Sanitize filename and prepare vectors for upsert | |
| sanitized_filename = re.sub(r'[^a-z0-9]', '-', original_filename.replace('.pdf', '').strip().lower()) | |
| sanitized_filename = re.sub(r'-+', '-', sanitized_filename).strip('-') | |
| vectors_to_upsert = [] | |
| for i, (doc, vec) in enumerate(zip(documents, embeddings)): | |
| vector_id = f"{sanitized_filename}-{i}" | |
| metadata = { | |
| "text": doc.page_content, | |
| "source": original_filename | |
| } | |
| vectors_to_upsert.append({"id": vector_id, "values": vec, "metadata": metadata}) | |
| # 4. Upsert vectors to Pinecone in batches | |
| batch_size = 100 | |
| for i in range(0, len(vectors_to_upsert), batch_size): | |
| batch = vectors_to_upsert[i:i+batch_size] | |
| index.upsert(vectors=batch) | |
| return f"File '{original_filename}' successfully embedded!", get_files_df() | |
| except Exception as e: | |
| return f"Unable to create embeddings: {str(e)}", get_files_df() | |
| # --- Gradio Interface Functions --- | |
| def get_files_df(): | |
| """Creates a DataFrame from the list of stored files for Gradio display.""" | |
| files = get_stored_files() | |
| if files: | |
| return pd.DataFrame({"Stored Files": files}) | |
| else: | |
| return pd.DataFrame({"Stored Files": []}) | |
| def handle_file_selection(evt: gr.SelectData): | |
| """Handles the file selection event from the DataFrame.""" | |
| if evt.value: | |
| return evt.value | |
| return "" | |
| # --- Gradio UI --- | |
| with gr.Blocks(theme=gr.themes.Soft(), title="PDF Uploader") as demo: | |
| gr.Markdown("# PDF File Uploader for Chatbot") | |
| gr.Markdown("Upload PDF files to add their content to the chatbot's knowledge base.") | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| gr.Markdown("## 📤 Upload New File") | |
| file_uploader = gr.File( | |
| label="Upload your PDF file", | |
| file_types=[".pdf"], | |
| type="filepath" | |
| ) | |
| upload_button = gr.Button("Upload to Chatbot Memory", variant="primary") | |
| upload_status = gr.Markdown("") | |
| with gr.Column(scale=1): | |
| gr.Markdown("## 🗂️ Stored Files") | |
| refresh_button = gr.Button("Refresh File List") | |
| file_df = gr.DataFrame( | |
| value=get_files_df, | |
| headers=["Stored Files"], | |
| interactive=True | |
| ) | |
| selected_file_text = gr.Textbox( | |
| label="Selected File", | |
| interactive=False, | |
| placeholder="Click on a file above to select it" | |
| ) | |
| delete_button = gr.Button("🗑️ Delete Selected File", variant="stop") | |
| delete_status = gr.Markdown("") | |
| # --- Event Handlers --- | |
| upload_button.click( | |
| fn=embedder, | |
| inputs=[file_uploader], | |
| outputs=[upload_status, file_df] | |
| ) | |
| refresh_button.click( | |
| fn=get_files_df, | |
| inputs=[], | |
| outputs=[file_df] | |
| ) | |
| file_df.select( | |
| fn=handle_file_selection, | |
| outputs=[selected_file_text] | |
| ) | |
| delete_button.click( | |
| fn=delete_file_from_vectorstore, | |
| inputs=[selected_file_text], | |
| outputs=[delete_status, file_df] | |
| ) | |
| if __name__ == "__main__": | |
| demo.launch(share=True) |