#!/usr/bin/env python3 """ Compare OCR (Space) extraction against PyMuPDF text extraction for fragrance book pages. Reads the page images and the literature_pages.jsonl extracted text, runs a sample of pages through the private PINO OCR Space, and reports: - character-level similarity - missing/extra formula lines - per-page diff summary Usage: export HF_TOKEN=... python scripts/compare_ocr_vs_pymupdf.py --pages-dir fragrance-research/pages \ --pages-jsonl fragrance-research/extracted/literature_pages.jsonl \ --space https://mattbitzesty-pino-ocr.hf.space \ --sample 20 --output ocr_vs_pymupdf.json """ from __future__ import annotations import argparse import json import os import re import sys from difflib import SequenceMatcher from pathlib import Path import requests def similarity(a: str, b: str) -> float: return SequenceMatcher(None, a, b).ratio() def clean_text(text: str) -> str: # collapse whitespace, strip, lowercase for comparison return re.sub(r"\s+", " ", text).strip().lower() def extract_formula_lines(text: str) -> list[str]: """Naive heuristic: lines that look like 'amount material' or 'material amount'.""" lines = [] for line in text.splitlines(): line = line.strip() if re.match(r"^\d+\s+[A-Za-z]", line) or re.match(r"^[A-Za-z].*\s\d+$", line): lines.append(line) return lines def ocr_image(space_url: str, token: str, image_path: Path, timeout: float = 120.0) -> str: url = f"{space_url.rstrip('/')}/ocr" with open(image_path, "rb") as f: response = requests.post( url, files={"file": (image_path.name, f, "image/png")}, headers={"Authorization": f"Bearer {token}"}, timeout=timeout, ) response.raise_for_status() return response.json().get("text", "") def normalize_source(source: str) -> str: """Normalize a full source title to the folder slug used by split_pdfs.py.""" s = source.lower() # Strip common metadata after "--" (Anna's Archive suffix) if " -- " in s: s = s.split(" -- ")[0] s = re.sub(r"[^a-z0-9]", "", s) return s def load_page_text(path: Path) -> dict[tuple[str, int], str]: records = {} for line in path.read_text().strip().splitlines(): rec = json.loads(line) source = normalize_source(rec.get("source", "")) records[(source, rec.get("page", 0))] = rec.get("text", "") return records def main() -> int: parser = argparse.ArgumentParser(description="Compare OCR Space vs PyMuPDF text extraction") parser.add_argument("--pages-dir", required=True, type=Path, help="Directory containing page images") parser.add_argument("--pages-jsonl", required=True, type=Path, help="literature_pages.jsonl") parser.add_argument("--space", default="https://mattbitzesty-pino-ocr.hf.space", help="OCR Space URL") parser.add_argument("--sample", type=int, default=20, help="Number of pages to sample") parser.add_argument("--output", type=Path, default=Path("data/ocr_vs_pymupdf.json"), help="Output JSON") parser.add_argument("--timeout", type=float, default=120.0, help="OCR request timeout") args = parser.parse_args() token = os.environ.get("HF_TOKEN") if not token: print("Set HF_TOKEN", file=sys.stderr) return 1 page_text = load_page_text(args.pages_jsonl) # Collect images that correspond to known pages images = sorted(args.pages_dir.rglob("*.png")) # Map image path -> (source, page) candidates = [] folder_to_source = {normalize_source(d.name): d.name for d in args.pages_dir.iterdir() if d.is_dir()} for img in images: # Expect page_0123.png m = re.search(r"page_(\d+)", img.name) if not m: continue page = int(m.group(1)) folder = normalize_source(img.parent.name) if (folder, page) in page_text: raw_folder = folder_to_source.get(folder, folder) candidates.append((folder, raw_folder, page, img)) if not candidates: print(f"No matching images found in {args.pages_dir}", file=sys.stderr) return 1 # Sample across books import random random.seed(42) sample = random.sample(candidates, min(args.sample, len(candidates))) results = [] for folder, source, page, img in sample: manual = page_text[(folder, page)] try: ocr_text = ocr_image(args.space, token, img, timeout=args.timeout) except Exception as exc: ocr_text = "" print(f"OCR failed for {source} page {page}: {exc}", file=sys.stderr) sim = similarity(clean_text(manual), clean_text(ocr_text)) manual_formulas = extract_formula_lines(manual) ocr_formulas = extract_formula_lines(ocr_text) results.append({ "source": source, "page": page, "image": str(img), "similarity": sim, "manual_chars": len(manual), "ocr_chars": len(ocr_text), "manual_formula_lines": manual_formulas, "ocr_formula_lines": ocr_formulas, }) print(f"{source} p{page}: sim={sim:.2f} manual={len(manual)} ocr={len(ocr_text)}") args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text(json.dumps(results, indent=2, ensure_ascii=False)) print(f"Wrote {len(results)} comparisons to {args.output}") return 0 if __name__ == "__main__": raise SystemExit(main())