themis / phase1 /scripts /document_text.py
vg15o2's picture
Moonley backend (HF Space build)
1d9bd9b
Raw
History Blame Contribute Delete
3 kB
"""Private document text extraction with OCR fallback for scanned PDFs."""
from __future__ import annotations
import re
from dataclasses import asdict, dataclass
from pathlib import Path
@dataclass(frozen=True)
class ExtractedDocument:
text: str
method: str
pages: int
ocr_pages: int
truncated: bool
def public_dict(self) -> dict:
value = asdict(self)
value.pop("text", None)
value["text_chars"] = len(self.text)
return value
def _normalise(text: object) -> str:
value = str(text or "").replace("\x00", "")
value = re.sub(r"[ \t]+", " ", value)
value = re.sub(r"\n{3,}", "\n\n", value)
return value.strip()
def _trim(text: str, max_chars: int) -> tuple[str, bool]:
if len(text) <= max_chars:
return text, False
return text[:max_chars].rsplit(" ", 1)[0].rstrip() + "\n\n[Document truncated]", True
def extract_document(path: str | Path, media_type: str = "", *, max_chars: int = 120_000) -> ExtractedDocument:
file_path = Path(path)
suffix = file_path.suffix.lower()
if suffix in {".txt", ".md"}:
text, truncated = _trim(_normalise(file_path.read_text(encoding="utf-8-sig")), max_chars)
return ExtractedDocument(text, "text", 1, 0, truncated)
if suffix == ".docx":
from docx import Document
document = Document(str(file_path))
blocks = [paragraph.text for paragraph in document.paragraphs if paragraph.text.strip()]
for table in document.tables:
for row in table.rows:
blocks.append(" | ".join(cell.text.strip() for cell in row.cells))
text, truncated = _trim(_normalise("\n\n".join(blocks)), max_chars)
return ExtractedDocument(text, "docx", 1, 0, truncated)
if suffix != ".pdf" and media_type != "application/pdf":
raise ValueError("Unsupported document type for text extraction.")
import pymupdf as fitz
pages: list[str] = []
ocr_pages = 0
with fitz.open(str(file_path)) as pdf:
page_count = len(pdf)
for page in pdf:
value = _normalise(page.get_text("text"))
if len(value) < 40:
try:
import io
import pytesseract
from PIL import Image
pixmap = page.get_pixmap(matrix=fitz.Matrix(2, 2), alpha=False)
image = Image.open(io.BytesIO(pixmap.tobytes("png")))
ocr = _normalise(pytesseract.image_to_string(image, lang="eng"))
if len(ocr) > len(value):
value = ocr
ocr_pages += 1
except (ImportError, OSError, RuntimeError):
pass
if value:
pages.append(value)
text, truncated = _trim(_normalise("\n\n".join(pages)), max_chars)
method = "pdf+ocr" if ocr_pages else "pdf-text"
return ExtractedDocument(text, method, page_count, ocr_pages, truncated)