import gradio as gr import os import time from dotenv import load_dotenv from pinecone import Pinecone, ServerlessSpec from langchain_pinecone import PineconeVectorStore from langchain_openai import OpenAIEmbeddings from langchain_core.documents import Document from langchain_community.document_loaders import PyPDFLoader from langchain_text_splitters import RecursiveCharacterTextSplitter import io import pandas as pd import re # Make sure to add this import at the top of your file # Load environment variables from a .env file load_dotenv() # --- Backend Functions --- def get_stored_files(): """Retrieve a list of files currently stored in the Pinecone index""" try: pc = Pinecone(api_key=os.environ.get("PINECONE_API_KEY")) index_name = os.environ.get("PINECONE_INDEX_NAME") existing_indexes = [index_info["name"] for index_info in pc.list_indexes()] if index_name not in existing_indexes: return [] index = pc.Index(index_name) # Query with a dummy vector to fetch metadata. Increase top_k if you have more files. results = index.query(vector=[0.0] * 3072, top_k=10000, include_metadata=True) unique_files = set() if results.matches: for match in results.matches: if hasattr(match, 'metadata') and match.metadata and 'source' in match.metadata: unique_files.add(match.metadata['source']) return sorted(list(unique_files)) except Exception as e: print(f"Error retrieving stored files: {str(e)}") return [] def delete_file_from_vectorstore(filename): """Deletes all vectors associated with a specific filename from Pinecone.""" if not filename: return "No file selected for deletion.", get_files_df() try: pc = Pinecone(api_key=os.environ.get("PINECONE_API_KEY")) index_name = os.environ.get("PINECONE_INDEX_NAME") index = pc.Index(index_name) index.delete(filter={"source": {"$eq": filename}}) return f"Successfully deleted {filename}.", get_files_df() except Exception as e: return f"Error while deleting the file: {str(e)}", get_files_df() def embedder(uploaded_file_path): """Handles embedding of the uploaded PDF file.""" if uploaded_file_path is None: return "No file uploaded. Please upload a PDF.", get_files_df() try: original_filename = os.path.basename(uploaded_file_path) pc = Pinecone(api_key=os.environ.get("PINECONE_API_KEY")) index_name = os.environ.get("PINECONE_INDEX_NAME") existing_indexes = [index_info["name"] for index_info in pc.list_indexes()] if index_name not in existing_indexes: pc.create_index( name=index_name, dimension=3072, metric="cosine", spec=ServerlessSpec(cloud="aws", region="us-east-1"), ) while not pc.describe_index(index_name).status["ready"]: time.sleep(1) index = pc.Index(index_name) embeddings = OpenAIEmbeddings(model="text-embedding-3-large", api_key=os.environ.get("OPENAI_API_KEY")) vector_store = PineconeVectorStore(index=index, embedding=embeddings) loader = PyPDFLoader(uploaded_file_path) raw_documents = loader.load() for doc in raw_documents: doc.metadata['source'] = original_filename text_splitter = RecursiveCharacterTextSplitter( chunk_size=800, chunk_overlap=400, length_function=len, is_separator_regex=False, ) documents = text_splitter.split_documents(raw_documents) # --- INIZIO: CODICE CORRETTO --- # 1. Rimuove l'estensione, gli spazi iniziali/finali e converte in minuscolo. sanitized_filename = original_filename.replace('.pdf', '').strip().lower() # 2. Sostituisce qualsiasi carattere non alfanumerico con un trattino. sanitized_filename = re.sub(r'[^a-z0-9]', '-', sanitized_filename) # 3. Sostituisce trattini multipli consecutivi con un singolo trattino. sanitized_filename = re.sub(r'-+', '-', sanitized_filename) # 4. (Opzionale ma consigliato) Rimuove eventuali trattini all'inizio o alla fine. sanitized_filename = sanitized_filename.strip('-') # 5. Usa il nome sanificato per generare gli ID. uuids = [f"{sanitized_filename}-{i}" for i in range(len(documents))] # --- FINE: CODICE CORRETTO --- batch_size = 100 for i in range(0, len(documents), batch_size): batch_docs = documents[i:i+batch_size] batch_ids = uuids[i:i+batch_size] vector_store.add_documents(documents=batch_docs, ids=batch_ids) return f"File '{original_filename}' successfully embedded!", get_files_df() except Exception as e: return f"Unable to create embeddings: {str(e)}", get_files_df() # --- Gradio Interface Functions --- def get_files_df(): files = get_stored_files() if files: return pd.DataFrame({"Stored Files": files}) else: return pd.DataFrame({"Stored Files": []}) # CORRECTION: Updated function to handle the select event correctly def handle_file_selection(evt: gr.SelectData): """ Handles the file selection event from the DataFrame. evt.value contains the value of the selected cell. """ if evt.value: return evt.value return "" # --- Gradio UI --- with gr.Blocks(theme=gr.themes.Soft(), title="PDF Uploader") as demo: gr.Markdown("# PDF File Uploader for Chatbot") gr.Markdown("Upload PDF files to add their content to the chatbot's knowledge base.") with gr.Row(): with gr.Column(scale=1): gr.Markdown("## 📤 Upload New File") file_uploader = gr.File( label="Upload your PDF file", file_types=[".pdf"], type="filepath" ) upload_button = gr.Button("Upload to Chatbot Memory", variant="primary") upload_status = gr.Markdown("") with gr.Column(scale=1): gr.Markdown("## 🗂️ Stored Files") refresh_button = gr.Button("Refresh File List") file_df = gr.DataFrame( value=get_files_df, headers=["Stored Files"], interactive=True ) selected_file_text = gr.Textbox( label="Selected File", interactive=False, placeholder="Click on a file above to select it" ) delete_button = gr.Button("🗑️ Delete Selected File", variant="stop") delete_status = gr.Markdown("") # --- Event Handlers --- upload_button.click( fn=embedder, inputs=[file_uploader], outputs=[upload_status, file_df] ) refresh_button.click( fn=get_files_df, inputs=[], outputs=[file_df] ) # CORRECTION: Removed the 'inputs' argument. # The event data 'evt' is now passed automatically to 'handle_file_selection'. file_df.select( fn=handle_file_selection, inputs=None, # Explicitly setting to None or removing this line works outputs=[selected_file_text] ) delete_button.click( fn=delete_file_from_vectorstore, inputs=[selected_file_text], outputs=[delete_status, file_df] ) if __name__ == "__main__": demo.launch()