Spaces:
Sleeping
Sleeping
| import pandas as pd | |
| from bs4 import BeautifulSoup | |
| from pathlib import Path | |
| from langchain.text_splitter import RecursiveCharacterTextSplitter | |
| from sentence_transformers import SentenceTransformer | |
| import numpy as np | |
| import faiss | |
| import tiktoken | |
| CHUNK_SIZE = 300 | |
| CHUNK_OVERLAP = 50 | |
| EMBEDDING_MODEL = "all-MiniLM-L6-v2" | |
| TOP_K = 5 | |
| def tiktoken_len(text): | |
| return len(tiktoken.encoding_for_model("gpt-4").encode(text)) | |
| def load_html_docs(csv_path, html_dir=None): | |
| df = pd.read_csv(csv_path) | |
| base_path = Path(csv_path).parent if html_dir is None else Path(html_dir) | |
| documents = [] | |
| for _, row in df.iterrows(): | |
| html_path = base_path / row['File'] | |
| if not html_path.exists(): | |
| print(f"[ADVERTENCIA] Archivo no encontrado: {html_path}") | |
| continue | |
| with open(html_path, 'r', encoding='utf-8') as f: | |
| soup = BeautifulSoup(f.read(), 'html.parser') | |
| text = soup.get_text(separator="\n") | |
| print(f"[INFO] Cargando HTML: {html_path.name}") | |
| documents.append({ | |
| "title": row.get("Title", "Sin título"), | |
| "author_id": row.get("Author ID", "Desconocido"), | |
| "filename": html_path.name, | |
| "content": text | |
| }) | |
| return documents | |
| def process_documents(documents, chunk_size=CHUNK_SIZE, chunk_overlap=CHUNK_OVERLAP): | |
| splitter = RecursiveCharacterTextSplitter( | |
| chunk_size=chunk_size, | |
| chunk_overlap=chunk_overlap, | |
| length_function=tiktoken_len | |
| ) | |
| all_chunks = [] | |
| for doc in documents: | |
| chunks = splitter.create_documents([doc["content"]]) | |
| for chunk in chunks: | |
| chunk.metadata = doc | |
| all_chunks.extend(chunks) | |
| texts = [chunk.page_content for chunk in all_chunks] | |
| embedder = SentenceTransformer(EMBEDDING_MODEL) | |
| vectors = embedder.encode(texts) | |
| return vectors, all_chunks, embedder | |
| def build_faiss_index(vectors, chunks): | |
| if len(vectors) == 0: | |
| raise ValueError("No se generaron vectores. Verifica si los HTML contienen texto útil y fueron encontrados.") | |
| dim = vectors[0].shape[0] | |
| index = faiss.IndexFlatL2(dim) | |
| index.add(np.array(vectors)) | |
| chunk_lookup = {i: chunks[i] for i in range(len(chunks))} | |
| return index, chunk_lookup | |
| def get_context(query, embedder, index, chunk_lookup, top_k=5): | |
| q_vector = embedder.encode([query]) | |
| scores, indices = index.search(np.array(q_vector), top_k) | |
| selected = [chunk_lookup[i] for i in indices[0]] | |
| return "\n\n".join([doc.page_content for doc in selected]) |