pino-source-code / scripts /compare_ocr_vs_pymupdf.py
mattbitzesty's picture
feat(pino): literature seed integration, stratified dataset split, registry enrichment, and Space sync
31ee18f unverified
Raw
History Blame Contribute Delete
5.55 kB
#!/usr/bin/env python3
"""
Compare OCR (Space) extraction against PyMuPDF text extraction for fragrance book pages.
Reads the page images and the literature_pages.jsonl extracted text, runs a sample of
pages through the private PINO OCR Space, and reports:
- character-level similarity
- missing/extra formula lines
- per-page diff summary
Usage:
export HF_TOKEN=...
python scripts/compare_ocr_vs_pymupdf.py --pages-dir fragrance-research/pages \
--pages-jsonl fragrance-research/extracted/literature_pages.jsonl \
--space https://mattbitzesty-pino-ocr.hf.space \
--sample 20 --output ocr_vs_pymupdf.json
"""
from __future__ import annotations
import argparse
import json
import os
import re
import sys
from difflib import SequenceMatcher
from pathlib import Path
import requests
def similarity(a: str, b: str) -> float:
return SequenceMatcher(None, a, b).ratio()
def clean_text(text: str) -> str:
# collapse whitespace, strip, lowercase for comparison
return re.sub(r"\s+", " ", text).strip().lower()
def extract_formula_lines(text: str) -> list[str]:
"""Naive heuristic: lines that look like 'amount material' or 'material amount'."""
lines = []
for line in text.splitlines():
line = line.strip()
if re.match(r"^\d+\s+[A-Za-z]", line) or re.match(r"^[A-Za-z].*\s\d+$", line):
lines.append(line)
return lines
def ocr_image(space_url: str, token: str, image_path: Path, timeout: float = 120.0) -> str:
url = f"{space_url.rstrip('/')}/ocr"
with open(image_path, "rb") as f:
response = requests.post(
url,
files={"file": (image_path.name, f, "image/png")},
headers={"Authorization": f"Bearer {token}"},
timeout=timeout,
)
response.raise_for_status()
return response.json().get("text", "")
def normalize_source(source: str) -> str:
"""Normalize a full source title to the folder slug used by split_pdfs.py."""
s = source.lower()
# Strip common metadata after "--" (Anna's Archive suffix)
if " -- " in s:
s = s.split(" -- ")[0]
s = re.sub(r"[^a-z0-9]", "", s)
return s
def load_page_text(path: Path) -> dict[tuple[str, int], str]:
records = {}
for line in path.read_text().strip().splitlines():
rec = json.loads(line)
source = normalize_source(rec.get("source", ""))
records[(source, rec.get("page", 0))] = rec.get("text", "")
return records
def main() -> int:
parser = argparse.ArgumentParser(description="Compare OCR Space vs PyMuPDF text extraction")
parser.add_argument("--pages-dir", required=True, type=Path, help="Directory containing page images")
parser.add_argument("--pages-jsonl", required=True, type=Path, help="literature_pages.jsonl")
parser.add_argument("--space", default="https://mattbitzesty-pino-ocr.hf.space", help="OCR Space URL")
parser.add_argument("--sample", type=int, default=20, help="Number of pages to sample")
parser.add_argument("--output", type=Path, default=Path("data/ocr_vs_pymupdf.json"), help="Output JSON")
parser.add_argument("--timeout", type=float, default=120.0, help="OCR request timeout")
args = parser.parse_args()
token = os.environ.get("HF_TOKEN")
if not token:
print("Set HF_TOKEN", file=sys.stderr)
return 1
page_text = load_page_text(args.pages_jsonl)
# Collect images that correspond to known pages
images = sorted(args.pages_dir.rglob("*.png"))
# Map image path -> (source, page)
candidates = []
folder_to_source = {normalize_source(d.name): d.name for d in args.pages_dir.iterdir() if d.is_dir()}
for img in images:
# Expect page_0123.png
m = re.search(r"page_(\d+)", img.name)
if not m:
continue
page = int(m.group(1))
folder = normalize_source(img.parent.name)
if (folder, page) in page_text:
raw_folder = folder_to_source.get(folder, folder)
candidates.append((folder, raw_folder, page, img))
if not candidates:
print(f"No matching images found in {args.pages_dir}", file=sys.stderr)
return 1
# Sample across books
import random
random.seed(42)
sample = random.sample(candidates, min(args.sample, len(candidates)))
results = []
for folder, source, page, img in sample:
manual = page_text[(folder, page)]
try:
ocr_text = ocr_image(args.space, token, img, timeout=args.timeout)
except Exception as exc:
ocr_text = ""
print(f"OCR failed for {source} page {page}: {exc}", file=sys.stderr)
sim = similarity(clean_text(manual), clean_text(ocr_text))
manual_formulas = extract_formula_lines(manual)
ocr_formulas = extract_formula_lines(ocr_text)
results.append({
"source": source,
"page": page,
"image": str(img),
"similarity": sim,
"manual_chars": len(manual),
"ocr_chars": len(ocr_text),
"manual_formula_lines": manual_formulas,
"ocr_formula_lines": ocr_formulas,
})
print(f"{source} p{page}: sim={sim:.2f} manual={len(manual)} ocr={len(ocr_text)}")
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(results, indent=2, ensure_ascii=False))
print(f"Wrote {len(results)} comparisons to {args.output}")
return 0
if __name__ == "__main__":
raise SystemExit(main())