Spaces:
Running on Zero
Running on Zero
Download knowledge_base.py from stevafernandes/chatbot-1: direct link, hf CLI and curl.
- Browser
- Download file 20.7 kB
-
https://huggingface.co/spaces/stevafernandes/chatbot-1/resolve/main/knowledge_base.py
- Command line
-
hf download hf://spaces/stevafernandes/chatbot-1/knowledge_base.py
-
curl -L -o knowledge_base.py https://huggingface.co/spaces/stevafernandes/chatbot-1/resolve/main/knowledge_base.py
20.7 kB
| """ | |
| knowledge_base.py — Retrieval-Augmented Generation over the uploaded ZIP corpus. | |
| At import/startup: | |
| 1. Locate the corpus (an extracted folder, or a .zip we extract once). | |
| 2. Extract text from every PDF (poppler pdftotext -> pypdf fallback) and | |
| OCR any PNG/JPG (optional; skipped gracefully if tesseract is absent). | |
| 3. Chunk the text, embed all chunks once with Gemini embeddings, cache to disk. | |
| At query time: | |
| - embed the user question, cosine-retrieve top-K chunks, and return them so | |
| the chat model can answer STRICTLY from that context. | |
| """ | |
| import os, re, glob, json, hashlib, subprocess, zipfile, shutil, time | |
| import numpy as np | |
| from google.genai import types | |
| EMBED_MODEL = "gemini-embedding-001" | |
| EMBED_DIM = 768 # output_dimensionality for the embedding model | |
| TOP_K = 8 # chunks retrieved per query | |
| TARGET_CHARS = 1400 | |
| OVERLAP = 200 | |
| MACOSX = "__MACOSX" | |
| # Where to look for the corpus, in priority order. On HF Spaces the repo root | |
| # is the app's working dir, so an uploaded "Archive_2.zip" or a pre-extracted | |
| # "knowledge_base/" folder both work. | |
| def _corpus_candidates(): | |
| return [os.environ.get("KB_DIR", "").strip(), "knowledge_base", "kb", "data", "."] | |
| def _zip_candidates(): | |
| return [os.environ.get("KB_ZIP", "").strip(), "Archive_2.zip"] | |
| EXTRACT_DIR = "_kb_extracted" | |
| CACHE_PATH = "_kb_index.npz" | |
| MANIFEST_PATH = "_kb_manifest.json" | |
| # ---------------------------------------------------------------------- | |
| # Corpus location / extraction | |
| # ---------------------------------------------------------------------- | |
| def _has_docs(d): | |
| if not d or not os.path.isdir(d): | |
| return False | |
| for p in glob.glob(os.path.join(d, "**", "*"), recursive=True): | |
| if MACOSX in p or os.path.basename(p).startswith("._"): | |
| continue | |
| if p.lower().endswith((".pdf", ".png", ".jpg", ".jpeg", ".txt", ".md")): | |
| return True | |
| return False | |
| def locate_corpus(): | |
| """Return a directory containing the source documents, extracting a zip if needed.""" | |
| for d in _corpus_candidates(): | |
| if _has_docs(d): | |
| # Avoid treating '.' as corpus if it only contains app files; | |
| # require at least one PDF for the '.' case. | |
| if d == "." and not glob.glob("*.pdf"): | |
| continue | |
| return d | |
| for z in _zip_candidates(): | |
| if z and os.path.isfile(z): | |
| os.makedirs(EXTRACT_DIR, exist_ok=True) | |
| with zipfile.ZipFile(z) as zf: | |
| for name in zf.namelist(): | |
| if MACOSX in name or os.path.basename(name).startswith("._"): | |
| continue | |
| zf.extract(name, EXTRACT_DIR) | |
| if _has_docs(EXTRACT_DIR): | |
| return EXTRACT_DIR | |
| return None | |
| def list_source_files(kb_dir): | |
| out = [] | |
| for p in sorted(glob.glob(os.path.join(kb_dir, "**", "*"), recursive=True)): | |
| if MACOSX in p or os.path.basename(p).startswith("._"): | |
| continue | |
| if os.path.isfile(p) and p.lower().endswith((".pdf", ".png", ".jpg", ".jpeg", ".txt", ".md")): | |
| out.append(p) | |
| return out | |
| # ---------------------------------------------------------------------- | |
| # Text extraction | |
| # ---------------------------------------------------------------------- | |
| def _pdftotext(path): | |
| try: | |
| r = subprocess.run(["pdftotext", "-layout", path, "-"], | |
| capture_output=True, timeout=120) | |
| if r.returncode == 0: | |
| return r.stdout.decode("utf-8", "ignore") | |
| except Exception: | |
| pass | |
| return "" | |
| def _pypdf_text(path): | |
| try: | |
| from pypdf import PdfReader | |
| reader = PdfReader(path) | |
| return "\n".join((pg.extract_text() or "") for pg in reader.pages) | |
| except Exception: | |
| return "" | |
| def _ocr_image(path): | |
| try: | |
| import pytesseract | |
| from PIL import Image | |
| return pytesseract.image_to_string(Image.open(path).convert("RGB")) | |
| except Exception: | |
| return "" | |
| def _ocr_pdf(path): | |
| """Rasterize + OCR a scanned/image PDF. Optional; empty if tools missing.""" | |
| try: | |
| import pytesseract | |
| from pdf2image import convert_from_path | |
| text = [] | |
| for img in convert_from_path(path, dpi=200): | |
| text.append(pytesseract.image_to_string(img)) | |
| return "\n".join(text) | |
| except Exception: | |
| return "" | |
| def extract_file_text(path): | |
| ext = os.path.splitext(path)[1].lower() | |
| if ext == ".pdf": | |
| txt = _pdftotext(path) | |
| if len(txt.strip()) < 50: | |
| txt = _pypdf_text(path) | |
| if len(txt.strip()) < 50: | |
| txt = _ocr_pdf(path) # image-only PDF | |
| return txt | |
| if ext in (".png", ".jpg", ".jpeg"): | |
| return _ocr_image(path) | |
| if ext in (".txt", ".md"): | |
| try: | |
| with open(path, encoding="utf-8", errors="ignore") as fh: | |
| return fh.read() | |
| except Exception: | |
| return "" | |
| return "" | |
| # ---------------------------------------------------------------------- | |
| # Chunking | |
| # ---------------------------------------------------------------------- | |
| def clean_text(t): | |
| t = t.replace("\x00", " ") | |
| t = re.sub(r"[ \t]+", " ", t) | |
| t = re.sub(r"\n{3,}", "\n\n", t) | |
| return t.strip() | |
| def chunk_text(text, source, target_chars=TARGET_CHARS, overlap=OVERLAP): | |
| text = clean_text(text) | |
| if not text: | |
| return [] | |
| chunks, start, n = [], 0, len(text) | |
| while start < n: | |
| end = min(start + target_chars, n) | |
| if end < n: | |
| window = text[start:end] | |
| cut = max(window.rfind("\n\n"), window.rfind(". "), window.rfind("\n")) | |
| if cut > target_chars * 0.5: | |
| end = start + cut + 1 | |
| chunk = text[start:end].strip() | |
| if len(chunk) > 40: | |
| chunks.append({"text": chunk, "source": os.path.basename(source)}) | |
| if end >= n: | |
| break | |
| start = max(end - overlap, start + 1) | |
| return chunks | |
| # ---------------------------------------------------------------------- | |
| # Source title formatting (filename -> human-readable document title) | |
| # ---------------------------------------------------------------------- | |
| def source_title(filename): | |
| """Human-readable title from a source filename. | |
| Only removes file extensions and export-timestamp suffixes and tidies | |
| separators — it never invents words, so it cannot hallucinate a title. If | |
| the cleaned result is a single token with no spaces (an accession/ID code | |
| such as 's41467-022-31430-0' or 'nihms385546'), it is prefixed with | |
| 'Document ' so the model refers to a document rather than a bare code. | |
| Always returns a non-empty string. | |
| """ | |
| if not filename: | |
| return "Reference document" | |
| original = os.path.basename(str(filename)) | |
| name = re.sub(r"\.(pdf|png|jpe?g|txt|md)$", "", original, flags=re.IGNORECASE) | |
| # strip trailing export/tracking timestamp suffix, e.g. "-07-13-2026_04_20_PM" | |
| name = re.sub(r"[-_]\d{1,2}[-_]\d{1,2}[-_]\d{2,4}[_-]\d{1,2}[_-]\d{2}[_-][AP]M$", "", name) | |
| # tidy separators | |
| name = name.replace(" _ ", " — ") # site/section separator used by several files | |
| name = name.replace("+", " ").replace("_", " ") | |
| # hyphens joining two letters -> spaces (leave digit-joining hyphens in IDs alone) | |
| name = re.sub(r"(?<=[A-Za-z])-(?=[A-Za-z])", " ", name) | |
| name = re.sub(r"\s{2,}", " ", name).strip(" -—") | |
| if not name: | |
| return original | |
| if " " not in name: # single token => identifier/code | |
| return f"Document {name}" | |
| return name | |
| # ---------------------------------------------------------------------- | |
| # Participant-facing source links | |
| # ---------------------------------------------------------------------- | |
| # Keyed by the exact filename used in the RAG corpus. | |
| # If a source has a public/participant-accessible URL, add it here. | |
| # If it does not, leave it out or set it to None; its citation will remain | |
| # plain text rather than clickable. | |
| SOURCE_REGISTRY = { | |
| "Autologous Stem Cell Transplant for Multiple Myeloma (IMF, 2026).pdf": | |
| "https://www.myeloma.org/autologous-stem-cell-transplant", | |
| "Bispecific Therapies for Multiple Myeloma (International Myeloma Foundation, 2026).pdf": | |
| "https://www.myeloma.org/emerging-therapies/bispecific-therapies", | |
| "Blood Protein Tests for Multiple Myeloma (IMF, 2025).pdf": | |
| "https://www.myeloma.org/blood-protein-testing", | |
| "Bone Marrow Tests for Multiple Myeloma (IMF, 2025).pdf": | |
| "https://www.myeloma.org/bone-marrow-tests", | |
| "CAR T-Cell Therapy for Multiple Myeloma (International Myeloma Foundation, 2024).pdf": | |
| "https://www.myeloma.org/emerging-therapies/car-t-cell-therapy", | |
| "Comprehensive Molecular Profiling of Multiple Myeloma by African and European Ancestry (Manojlovic et al., 2017).pdf": | |
| "https://journals.plos.org/plosgenetics/article?id=10.1371/journal.pgen.1007087", | |
| "Durie-Salmon Staging for Multiple Myeloma (IMF, 2026).pdf": | |
| "https://www.myeloma.org/durie-salmon-staging", | |
| "FDA Approval of Teclistamab Plus Daratumumab for Relapsed Multiple Myeloma (2026).pdf": | |
| "https://www.fda.gov/drugs/resources-information-approved-drugs/fda-approves-teclistamab-combination-daratumumab-hyaluronidase-fihj-relapsed-or-refractory-multiple", | |
| "First-Line Treatment for Multiple Myeloma (IMF, 2026).pdf": | |
| "https://www.myeloma.org/frontline-treatment-options", | |
| "Genetic Heterogeneity and Drug Resistance in Relapsed Multiple Myeloma (Vo et al., 2022).pdf": | |
| "https://www.nature.com/articles/s41467-022-31430-0", | |
| "Genetic Heterogeneity in Multiple Myeloma and Targeted Therapy (Lohr et al., 2014).pdf": | |
| "https://www.cell.com/cancer-cell/fulltext/S1535-6108(13)00542-4?_returnURL=https%3A%2F%2Flinkinghub.elsevier.com%2Fretrieve%2Fpii%2FS1535610813005424%3Fshowall%3Dtrue", | |
| "Immunocytokine Therapy for Multiple Myeloma (International Myeloma Foundation, 2026).pdf": | |
| "https://www.myeloma.org/emerging-therapies/immunocytokine-therapy", | |
| "Initial Genome Sequencing of Multiple Myeloma (Chapman et al., 2011).pdf": | |
| "https://www.nature.com/articles/nature09837", | |
| "ISS and R-ISS Staging for Multiple Myeloma (IMF, 2026).pdf": | |
| "https://www.myeloma.org/international-staging-system-iss-reivised-iss-r-iss", | |
| "Kidney Function Tests for Multiple Myeloma (IMF, 2025).pdf": | |
| "https://www.myeloma.org/myeloma-kidney-tests", | |
| "M-Protein Tests for Multiple Myeloma (IMF, 2026).pdf": | |
| "https://www.myeloma.org/monoclonal-protein-tests", | |
| "Maintenance Therapy for Multiple Myeloma (IMF, 2024).pdf": | |
| "https://www.myeloma.org/maintenance-therapy", | |
| "Mayo Clinic mSMART Relapsed Myeloma Treatment Guidelines, Version 9 (2026).pdf": | |
| "https://static1.squarespace.com/static/5b44f08ac258b493a25098a3/t/69b45653e3e1365b6857a106/1773426259531/03_13_26_v9+Relapsed+MM+mSMART%2BRelapsedMM.pdf", | |
| "MGUS, Smoldering Myeloma, and Active Myeloma (IMF, 2026).pdf": | |
| "https://www.myeloma.org/what-are-mgus-smm-mm", | |
| "Minimal Residual Disease in Multiple Myeloma (IMF).pdf": | |
| "https://www.myeloma.org/videos/minimal-residual-disease-mrd", | |
| "Multiple Myeloma Autologous Stem Cell Transplantation Patient Toolkit (MMRF, 2026).pdf": | |
| "https://themmrf.org/wp-content/uploads/2026/03/Autologous-Stem-Cell-Transplantation-patient_toolkit-2026.pdf", | |
| "Multiple Myeloma Blood Tests (IMF, 2026).pdf": | |
| "https://www.myeloma.org/multiple-myeloma-blood-tests", | |
| "Multiple Myeloma Complications (IMF, 2025).pdf": | |
| "https://www.myeloma.org/medical-conditions", | |
| "Multiple Myeloma Diagnosis and Treatment (Mayo Clinic, 2024).pdf": | |
| "https://www.mayoclinic.org/diseases-conditions/multiple-myeloma/diagnosis-treatment/drc-20353383", | |
| "Multiple Myeloma FAQ (IMF, 2026).pdf": | |
| "https://www.myeloma.org/myeloma-cancer-questions", | |
| "Multiple Myeloma Imaging Studies (IMF, 2025).pdf": | |
| "https://www.myeloma.org/multiple-myeloma-imaging-studies", | |
| "Multiple Myeloma Immunotherapy Patient Toolkit (MMRF, 2026).pdf": | |
| "https://themmrf.org/wp-content/uploads/2026/03/Immunotherapy-patient_toolkit-2026.pdf", | |
| "Multiple Myeloma Overview (IMF, 2026).pdf": | |
| "https://www.myeloma.org/what-is-multiple-myeloma", | |
| "Multiple Myeloma Relapse (IMF, 2025).pdf": | |
| "https://www.myeloma.org/treatment/relapse-definition", | |
| "Multiple Myeloma Signs and Diagnosis (IMF, 2025).pdf": | |
| "https://www.myeloma.org/newly-diagnosed/do-you-have-myeloma", | |
| "Multiple Myeloma Staging and Risk Stratification (IMF, 2019).pdf": | |
| "https://www.myeloma.org/multiple-myeloma/multiple-myeloma-prognosis-staging", | |
| "Multiple Myeloma Testing (IMF, 2026).pdf": | |
| "https://www.myeloma.org/multiple-myeloma-tests", | |
| "Multiple Myeloma Testing, Staging and Prognosis (IMF, 2025).pdf": | |
| "https://www.myeloma.org/tests-staging", | |
| "NCCN Multiple Myeloma Patient Guidelines (2026).pdf": | |
| "https://www.nccn.org/patients/guidelines/content/PDF/myeloma-patient.pdf", | |
| "Perspectives on the Risk-Stratified Treatment of Multiple Myeloma (Davies et al., 2022).pdf": | |
| "https://pmc.ncbi.nlm.nih.gov/articles/PMC9894570/", | |
| "Quality of Life in Longitudinal Multiple Myeloma Studies (Nielsen et al., 2017).pdf": | |
| "https://onlinelibrary.wiley.com/doi/10.1111/ejh.12882", | |
| "Quality of Life in Multiple Myeloma Treatment (Sonneveld et al., 2013).pdf": | |
| "https://www.nature.com/articles/leu2013185", | |
| "Questions to Ask Your Myeloma Doctor (IMF).pdf": | |
| "https://www.myeloma.org/resource-library/tip-card-ask-your-doctor", | |
| "Routine Health Screenings for Myeloma Patients (IMF, 2021).pdf": | |
| "https://www.myeloma.org/health-screenings", | |
| "The Role of Minimal Residual Disease Testing in Myeloma Treatment Selection and Drug Development (Anderson et al., 2017).pdf": | |
| "https://aacrjournals.org/clincancerres/article/23/15/3980/257045/The-Role-of-Minimal-Residual-Disease-Testing-in", | |
| "Types of Multiple Myeloma (IMF, 2025).pdf": | |
| "https://www.myeloma.org/types-of-myeloma", | |
| } | |
| # ---------------------------------------------------------------------- | |
| # Embedding (Gemini) with batching + disk cache | |
| # ---------------------------------------------------------------------- | |
| def _l2norm(v): | |
| v = np.asarray(v, dtype=np.float32) | |
| nrm = np.linalg.norm(v, axis=-1, keepdims=True) | |
| nrm[nrm == 0] = 1.0 | |
| return v / nrm | |
| def _embed_batch(client, texts, task_type): | |
| """Return np.array (len(texts), EMBED_DIM). Falls back to per-item on error.""" | |
| cfg = types.EmbedContentConfig(task_type=task_type, output_dimensionality=EMBED_DIM) | |
| try: | |
| resp = client.models.embed_content(model=EMBED_MODEL, contents=texts, config=cfg) | |
| return np.array([e.values for e in resp.embeddings], dtype=np.float32) | |
| except Exception: | |
| vecs = [] | |
| for t in texts: | |
| resp = client.models.embed_content(model=EMBED_MODEL, contents=t, config=cfg) | |
| vecs.append(resp.embeddings[0].values) | |
| return np.array(vecs, dtype=np.float32) | |
| def _corpus_fingerprint(files): | |
| h = hashlib.sha256() | |
| for p in sorted(files): | |
| st = os.stat(p) | |
| h.update(p.encode()); h.update(str(st.st_size).encode()); h.update(str(int(st.st_mtime)).encode()) | |
| h.update(f"{EMBED_MODEL}:{EMBED_DIM}:{TARGET_CHARS}:{OVERLAP}".encode()) | |
| return h.hexdigest() | |
| class KnowledgeBase: | |
| def __init__(self): | |
| self.ready = False | |
| self.error = None | |
| self.chunks = [] # list of {text, source} | |
| self.matrix = None # (N, EMBED_DIM) normalized | |
| self.sources = [] # unique source filenames | |
| self.dir = None | |
| # ---- build or load ---- | |
| def build(self, client, embed_batch=_embed_batch, log=print): | |
| try: | |
| self.dir = locate_corpus() | |
| if not self.dir: | |
| self.error = ("No knowledge-base documents found. Upload Archive_2.zip to the " | |
| "Space (repo root) or add a knowledge_base/ folder of PDFs.") | |
| return False | |
| files = list_source_files(self.dir) | |
| if not files: | |
| self.error = "Knowledge-base folder found but contains no readable documents." | |
| return False | |
| fp = _corpus_fingerprint(files) | |
| if self._load_cache(fp): | |
| self.ready = True | |
| log(f"[KB] Loaded cached index: {len(self.chunks)} chunks from {len(self.sources)} files.") | |
| return True | |
| # Extract + chunk | |
| all_chunks = [] | |
| for p in files: | |
| txt = extract_file_text(p) | |
| cs = chunk_text(txt, p) | |
| if cs: | |
| all_chunks.extend(cs) | |
| log(f"[KB] {os.path.basename(p)}: {len(cs)} chunks") | |
| else: | |
| log(f"[KB] {os.path.basename(p)}: no extractable text (skipped)") | |
| if not all_chunks: | |
| self.error = "Documents found but no text could be extracted from any of them." | |
| return False | |
| # Embed in batches | |
| texts = [c["text"] for c in all_chunks] | |
| vecs = [] | |
| B = 64 | |
| for i in range(0, len(texts), B): | |
| batch = texts[i:i + B] | |
| vecs.append(embed_batch(client, batch, "RETRIEVAL_DOCUMENT")) | |
| log(f"[KB] embedded {min(i+B, len(texts))}/{len(texts)} chunks") | |
| matrix = np.vstack(vecs) | |
| self.chunks = all_chunks | |
| self.matrix = _l2norm(matrix) | |
| self.sources = sorted({c["source"] for c in all_chunks}) | |
| self._save_cache(fp) | |
| self.ready = True | |
| log(f"[KB] Index built: {len(self.chunks)} chunks from {len(self.sources)} files.") | |
| return True | |
| except Exception as e: | |
| self.error = f"Failed to build knowledge base: {e}" | |
| return False | |
| def _save_cache(self, fp): | |
| try: | |
| np.savez_compressed(CACHE_PATH, matrix=self.matrix, | |
| texts=np.array([c["text"] for c in self.chunks], dtype=object), | |
| srcs=np.array([c["source"] for c in self.chunks], dtype=object), | |
| fp=np.array([fp])) | |
| with open(MANIFEST_PATH, "w") as fh: | |
| json.dump({"fingerprint": fp, "sources": self.sources, | |
| "n_chunks": len(self.chunks)}, fh, indent=2) | |
| except Exception: | |
| pass | |
| def _load_cache(self, fp): | |
| if not os.path.isfile(CACHE_PATH): | |
| return False | |
| try: | |
| d = np.load(CACHE_PATH, allow_pickle=True) | |
| if str(d["fp"][0]) != fp: | |
| return False | |
| self.matrix = d["matrix"].astype(np.float32) | |
| texts = list(d["texts"]); srcs = list(d["srcs"]) | |
| self.chunks = [{"text": t, "source": s} for t, s in zip(texts, srcs)] | |
| self.sources = sorted(set(srcs)) | |
| return True | |
| except Exception: | |
| return False | |
| def source_link_map(self): | |
| """ | |
| Map the human-readable document titles used in chatbot citations | |
| to participant-accessible URLs. | |
| Sources without a URL remain plain-text citations. | |
| """ | |
| links = {} | |
| for source in self.sources: | |
| url = SOURCE_REGISTRY.get(source) | |
| if url: | |
| links[source_title(source)] = url | |
| return links | |
| # ---- retrieve ---- | |
| def retrieve(self, client, query, k=TOP_K, embed_batch=_embed_batch): | |
| if not self.ready: | |
| return [] | |
| qv = embed_batch(client, [query], "RETRIEVAL_QUERY")[0] | |
| qv = _l2norm(qv) | |
| sims = self.matrix @ qv | |
| k = min(k, len(sims)) | |
| idx = np.argpartition(-sims, k - 1)[:k] | |
| idx = idx[np.argsort(-sims[idx])] | |
| return [{"text": self.chunks[i]["text"], "source": self.chunks[i]["source"], | |
| "score": float(sims[i])} for i in idx] | |
| def context_block(self, retrieved): | |
| """Format retrieved chunks as a source-attributed context block, labeling | |
| each passage with the document's readable title (not 'Source N').""" | |
| parts = [] | |
| for r in retrieved: | |
| title = source_title(r["source"]) | |
| parts.append(f"[Document: {title}]\n{r['text']}") | |
| return "\n\n---\n\n".join(parts) | |