Spaces:
Runtime error
Runtime error
| import os | |
| import mimetypes | |
| import faiss | |
| import hashlib | |
| import google.generativeai as genai | |
| from uuid import uuid4 | |
| from pdf2image import convert_from_path | |
| from google.cloud import vision | |
| from google.oauth2 import service_account | |
| from moviepy import VideoFileClip | |
| from langchain_core.documents import Document | |
| from langchain_community.vectorstores import FAISS | |
| from langchain_core.document_loaders.blob_loaders import Blob | |
| from langchain_community.docstore.in_memory import InMemoryDocstore | |
| from langchain_text_splitters import RecursiveCharacterTextSplitter | |
| from langchain_community.document_loaders import TextLoader, Docx2txtLoader | |
| from langchain_community.document_loaders.parsers import OpenAIWhisperParser | |
| from langchain_openai import OpenAIEmbeddings | |
| class VectorStoreManager: | |
| """ | |
| Clase para gestionar un vector store utilizando FAISS y otras herramientas avanzadas. | |
| Permite procesar archivos de texto, PDFs, imágenes, audios y videos. | |
| """ | |
| def __init__(self, openai_api_key, gemini_api_key, google_credentials_path, faiss_index_path="faiss_index"): | |
| """ | |
| Inicializa el gestor con las credenciales necesarias y configura los módulos requeridos. | |
| Args: | |
| openai_api_key (str): API key de OpenAI. | |
| gemini_api_key (str): API key de Gemini. | |
| google_credentials_path (str): Ruta al archivo de credenciales de Google. | |
| faiss_index_path (str): Ruta donde se guardará el índice FAISS. | |
| """ | |
| try: | |
| self.api_key = openai_api_key | |
| self.GOOGLE_CREDENTIALS_PATH = google_credentials_path | |
| self.GENAI_API_KEY = gemini_api_key | |
| genai.configure(api_key=self.GENAI_API_KEY) | |
| # Configuración de embeddings | |
| self.embeddings = OpenAIEmbeddings( | |
| api_key=self.api_key, | |
| model="text-embedding-3-large", | |
| max_retries=1, | |
| ) | |
| # Configuración de FAISS | |
| self.index = faiss.IndexFlatL2(len(self.embeddings.embed_query("test"))) | |
| self.vector_store = FAISS( | |
| embedding_function=self.embeddings, | |
| index=self.index, | |
| docstore=InMemoryDocstore(), | |
| index_to_docstore_id={}, | |
| ) | |
| self.faiss_index_path = faiss_index_path | |
| # Configuración del splitter | |
| self.text_splitter = RecursiveCharacterTextSplitter(chunk_size=2000, chunk_overlap=200) | |
| except Exception as e: | |
| raise RuntimeError(f"Error al inicializar VectorStoreManager: {e}") | |
| def generate_gemini_response(self, text): | |
| """ | |
| Corrige texto OCR utilizando un modelo generativo (Gemini). | |
| Args: | |
| text (str): Texto a corregir. | |
| Returns: | |
| str: Texto corregido. | |
| """ | |
| try: | |
| model = genai.GenerativeModel("gemini-1.5-flash") | |
| prompt = f""" | |
| Corrección profesional de texto extraído por OCR con los siguientes criterios: | |
| 1. Corregir errores de reconocimiento óptico de caracteres | |
| 2. Restaurar la puntuación y la estructura gramatical correcta | |
| 3. Conservar el formato original del documento | |
| 4. Mantener intacto el contenido informativo del texto original | |
| 5. Asegurar la máxima precisión en la transcripción | |
| Características de salida: | |
| - Texto completamente legible | |
| - Sin caracteres extraviados o mal reconocidos | |
| - Gramática y ortografía corregidas | |
| - Formato apropiado para archivo .txt | |
| Texto OCR: | |
| {text} | |
| Texto corregido: | |
| """ | |
| response = model.generate_content( | |
| prompt, | |
| generation_config=genai.GenerationConfig(max_output_tokens=4096, temperature=0.2, top_p=0.75) | |
| ) | |
| return response.text | |
| except Exception as e: | |
| raise RuntimeError(f"Error al generar respuesta de Gemini: {e}") | |
| def _calculate_file_hash(self, file_path): | |
| """ | |
| Calcula el hash SHA256 de un archivo. | |
| Args: | |
| file_path (str): Ruta del archivo. | |
| Returns: | |
| str: Hash SHA256 del archivo. | |
| """ | |
| try: | |
| hasher = hashlib.sha256() | |
| with open(file_path, 'rb') as f: | |
| while chunk := f.read(8192): | |
| hasher.update(chunk) | |
| return hasher.hexdigest() | |
| except Exception as e: | |
| raise RuntimeError(f"Error al calcular hash del archivo {file_path}: {e}") | |
| def save_vs_to_local(self): | |
| """ | |
| Guarda el vector store localmente, incluyendo el índice y el docstore. | |
| """ | |
| try: | |
| self.vector_store.save_local(self.faiss_index_path) | |
| print(f"Vector Store guardado en {self.faiss_index_path}.") | |
| except Exception as e: | |
| raise RuntimeError(f"Error al guardar el vector store: {e}") | |
| def load_vs_from_local(self): | |
| """ | |
| Carga el vector store desde el disco local. | |
| """ | |
| try: | |
| self.vector_store.load_local(self.faiss_index_path, self.embeddings, allow_dangerous_deserialization=True) | |
| print(f"Vector Store cargado desde {self.faiss_index_path}.") | |
| except Exception as e: | |
| raise RuntimeError(f"Error al cargar el vector store: {e}") | |
| def similarity_search(self, query, k=1): | |
| """ | |
| Realiza una búsqueda de similitud en el vector store. | |
| Args: | |
| query (str): Consulta. | |
| k (int): Número de resultados a retornar. | |
| Returns: | |
| List[Document]: Resultados de la búsqueda. | |
| """ | |
| try: | |
| return self.vector_store.similarity_search(query=query, k=k) | |
| except Exception as e: | |
| raise RuntimeError(f"Error en la búsqueda de similitud: {e}") | |
| def ocr_image(self, image_path): | |
| """ | |
| Realiza OCR en una imagen utilizando Google Vision API. | |
| Args: | |
| image_path (str): Ruta de la imagen. | |
| Returns: | |
| str: Texto extraído y corregido. | |
| """ | |
| try: | |
| credentials = service_account.Credentials.from_service_account_file(self.GOOGLE_CREDENTIALS_PATH) | |
| client = vision.ImageAnnotatorClient(credentials=credentials) | |
| with open(image_path, 'rb') as image_file: | |
| content = image_file.read() | |
| image = vision.Image(content=content) | |
| response = client.text_detection(image=image) | |
| if response.error.message: | |
| raise Exception(f"Error en Vision API: {response.error.message}") | |
| return self.generate_gemini_response(response.full_text_annotation.text) | |
| except Exception as e: | |
| raise RuntimeError(f"Error al realizar OCR en la imagen {image_path}: {e}") | |
| def process_pdf(self, file_path): | |
| """ | |
| Procesa un PDF dividiendo páginas en imágenes y aplicando OCR. | |
| Args: | |
| file_path (str): Ruta del archivo PDF. | |
| Returns: | |
| str: Texto extraído de todas las páginas. | |
| """ | |
| try: | |
| pages = convert_from_path(file_path) | |
| extracted_text = "" | |
| for page in pages: | |
| temp_image_path = "temp_page.jpg" | |
| page.save(temp_image_path, "JPEG") | |
| extracted_text += self.ocr_image(temp_image_path) + "\n" | |
| os.remove(temp_image_path) | |
| return extracted_text | |
| except Exception as e: | |
| raise RuntimeError(f"Error al procesar el PDF {file_path}: {e}") | |
| def process_audio(self, file_path): | |
| """ | |
| Transcribe un archivo de audio usando OpenAI Whisper. | |
| Args: | |
| file_path (str): Ruta del archivo de audio. | |
| Returns: | |
| str: Transcripción del audio. | |
| """ | |
| try: | |
| audio_blob = Blob.from_path(file_path) | |
| parser = OpenAIWhisperParser(api_key=self.api_key) | |
| documents = parser.parse(audio_blob) | |
| return "\n".join(doc.page_content for doc in documents) | |
| except Exception as e: | |
| raise RuntimeError(f"Error al procesar el audio {file_path}: {e}") | |
| def process_video(self, file_path): | |
| """ | |
| Extrae el audio de un video y lo transcribe. | |
| Args: | |
| file_path (str): Ruta del archivo de video. | |
| Returns: | |
| str: Transcripción del audio extraído. | |
| """ | |
| try: | |
| video = VideoFileClip(file_path) | |
| temp_audio_path = "temp_audio.wav" | |
| video.audio.write_audiofile(temp_audio_path) | |
| transcript = self.process_audio(temp_audio_path) | |
| os.remove(temp_audio_path) | |
| return transcript | |
| except Exception as e: | |
| raise RuntimeError(f"Error al procesar el video {file_path}: {e}") | |
| def load_directory(self, directory_path): | |
| """ | |
| Carga y procesa todos los archivos de un directorio de manera recursiva. | |
| Args: | |
| directory_path (str): Ruta del directorio. | |
| Raises: | |
| ValueError: Si la ruta proporcionada no es un directorio válido. | |
| """ | |
| try: | |
| if not os.path.isdir(directory_path): | |
| raise ValueError(f"{directory_path} no es un directorio válido.") | |
| for root, _, files in os.walk(directory_path): | |
| for file in files: | |
| file_path = os.path.join(root, file) | |
| try: | |
| print(f"Procesando archivo: {file_path}") | |
| self.load_file(file_path) | |
| except Exception as e: | |
| print(f"Error procesando el archivo {file_path}: {e}") | |
| print(f"Todos los archivos en el directorio '{directory_path}' han sido procesados.") | |
| except Exception as e: | |
| raise RuntimeError(f"Error al procesar el directorio {directory_path}: {e}") | |
| def load_file(self, file_path): | |
| """ | |
| Carga y procesa un archivo, dividiéndolo y agregándolo al vector store con metadatos. | |
| Args: | |
| file_path (str): Ruta del archivo. | |
| """ | |
| try: | |
| mime_type, _ = mimetypes.guess_type(file_path) | |
| if mime_type is None: | |
| raise ValueError(f"No se pudo determinar el tipo de archivo para: {file_path}") | |
| file_hash = self._calculate_file_hash(file_path) | |
| file_name = os.path.basename(file_path) | |
| # Verificar duplicados | |
| for doc_id, document in self.vector_store.docstore._dict.items(): | |
| if document.metadata.get("hash") == file_hash: | |
| print(f"Archivo '{file_name}' ya procesado. Omitiendo...") | |
| return | |
| # Procesar según el tipo de archivo | |
| documents = [] | |
| if mime_type.startswith("text"): | |
| loader = TextLoader(file_path, autodetect_encoding=True) | |
| raw_documents = loader.load() | |
| documents = [ | |
| Document(page_content=doc.page_content, metadata={"source": file_name, "hash": file_hash}) | |
| for doc in raw_documents | |
| ] | |
| elif mime_type in ["application/msword", | |
| "application/vnd.openxmlformats-officedocument.wordprocessingml.document"]: | |
| # Procesar archivos .doc o .docx | |
| loader = Docx2txtLoader(file_path) | |
| raw_documents = loader.load() | |
| documents = [ | |
| Document(page_content=doc.page_content, metadata={"source": file_name, "hash": file_hash}) | |
| for doc in raw_documents | |
| ] | |
| elif mime_type == "application/pdf": | |
| text = self.process_pdf(file_path) | |
| documents = [Document(page_content=text, metadata={"source": file_name, "hash": file_hash})] | |
| elif mime_type.startswith("image"): | |
| text = self.ocr_image(file_path) | |
| documents = [Document(page_content=text, metadata={"source": file_name, "hash": file_hash})] | |
| elif mime_type.startswith("audio"): | |
| text = self.process_audio(file_path) | |
| documents = [Document(page_content=text, metadata={"source": file_name, "hash": file_hash})] | |
| elif mime_type.startswith("video"): | |
| text = self.process_video(file_path) | |
| documents = [Document(page_content=text, metadata={"source": file_name, "hash": file_hash})] | |
| else: | |
| raise ValueError(f"Tipo de archivo no soportado: {mime_type}") | |
| # Dividir y agregar al vector store | |
| for doc in documents: | |
| chunks = self.text_splitter.split_documents([doc]) | |
| self.vector_store.add_documents(chunks) | |
| print(f"Archivo '{file_name}' procesado y agregado al vector store.") | |
| except Exception as e: | |
| raise RuntimeError(f"Error al cargar el archivo {file_path}: {e}") | |