feat(pino): literature seed integration, stratified dataset split, registry enrichment, and Space sync
31ee18f unverified | #!/usr/bin/env python3 | |
| """ | |
| Compare OCR (Space) extraction against PyMuPDF text extraction for fragrance book pages. | |
| Reads the page images and the literature_pages.jsonl extracted text, runs a sample of | |
| pages through the private PINO OCR Space, and reports: | |
| - character-level similarity | |
| - missing/extra formula lines | |
| - per-page diff summary | |
| Usage: | |
| export HF_TOKEN=... | |
| python scripts/compare_ocr_vs_pymupdf.py --pages-dir fragrance-research/pages \ | |
| --pages-jsonl fragrance-research/extracted/literature_pages.jsonl \ | |
| --space https://mattbitzesty-pino-ocr.hf.space \ | |
| --sample 20 --output ocr_vs_pymupdf.json | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| import os | |
| import re | |
| import sys | |
| from difflib import SequenceMatcher | |
| from pathlib import Path | |
| import requests | |
| def similarity(a: str, b: str) -> float: | |
| return SequenceMatcher(None, a, b).ratio() | |
| def clean_text(text: str) -> str: | |
| # collapse whitespace, strip, lowercase for comparison | |
| return re.sub(r"\s+", " ", text).strip().lower() | |
| def extract_formula_lines(text: str) -> list[str]: | |
| """Naive heuristic: lines that look like 'amount material' or 'material amount'.""" | |
| lines = [] | |
| for line in text.splitlines(): | |
| line = line.strip() | |
| if re.match(r"^\d+\s+[A-Za-z]", line) or re.match(r"^[A-Za-z].*\s\d+$", line): | |
| lines.append(line) | |
| return lines | |
| def ocr_image(space_url: str, token: str, image_path: Path, timeout: float = 120.0) -> str: | |
| url = f"{space_url.rstrip('/')}/ocr" | |
| with open(image_path, "rb") as f: | |
| response = requests.post( | |
| url, | |
| files={"file": (image_path.name, f, "image/png")}, | |
| headers={"Authorization": f"Bearer {token}"}, | |
| timeout=timeout, | |
| ) | |
| response.raise_for_status() | |
| return response.json().get("text", "") | |
| def normalize_source(source: str) -> str: | |
| """Normalize a full source title to the folder slug used by split_pdfs.py.""" | |
| s = source.lower() | |
| # Strip common metadata after "--" (Anna's Archive suffix) | |
| if " -- " in s: | |
| s = s.split(" -- ")[0] | |
| s = re.sub(r"[^a-z0-9]", "", s) | |
| return s | |
| def load_page_text(path: Path) -> dict[tuple[str, int], str]: | |
| records = {} | |
| for line in path.read_text().strip().splitlines(): | |
| rec = json.loads(line) | |
| source = normalize_source(rec.get("source", "")) | |
| records[(source, rec.get("page", 0))] = rec.get("text", "") | |
| return records | |
| def main() -> int: | |
| parser = argparse.ArgumentParser(description="Compare OCR Space vs PyMuPDF text extraction") | |
| parser.add_argument("--pages-dir", required=True, type=Path, help="Directory containing page images") | |
| parser.add_argument("--pages-jsonl", required=True, type=Path, help="literature_pages.jsonl") | |
| parser.add_argument("--space", default="https://mattbitzesty-pino-ocr.hf.space", help="OCR Space URL") | |
| parser.add_argument("--sample", type=int, default=20, help="Number of pages to sample") | |
| parser.add_argument("--output", type=Path, default=Path("data/ocr_vs_pymupdf.json"), help="Output JSON") | |
| parser.add_argument("--timeout", type=float, default=120.0, help="OCR request timeout") | |
| args = parser.parse_args() | |
| token = os.environ.get("HF_TOKEN") | |
| if not token: | |
| print("Set HF_TOKEN", file=sys.stderr) | |
| return 1 | |
| page_text = load_page_text(args.pages_jsonl) | |
| # Collect images that correspond to known pages | |
| images = sorted(args.pages_dir.rglob("*.png")) | |
| # Map image path -> (source, page) | |
| candidates = [] | |
| folder_to_source = {normalize_source(d.name): d.name for d in args.pages_dir.iterdir() if d.is_dir()} | |
| for img in images: | |
| # Expect page_0123.png | |
| m = re.search(r"page_(\d+)", img.name) | |
| if not m: | |
| continue | |
| page = int(m.group(1)) | |
| folder = normalize_source(img.parent.name) | |
| if (folder, page) in page_text: | |
| raw_folder = folder_to_source.get(folder, folder) | |
| candidates.append((folder, raw_folder, page, img)) | |
| if not candidates: | |
| print(f"No matching images found in {args.pages_dir}", file=sys.stderr) | |
| return 1 | |
| # Sample across books | |
| import random | |
| random.seed(42) | |
| sample = random.sample(candidates, min(args.sample, len(candidates))) | |
| results = [] | |
| for folder, source, page, img in sample: | |
| manual = page_text[(folder, page)] | |
| try: | |
| ocr_text = ocr_image(args.space, token, img, timeout=args.timeout) | |
| except Exception as exc: | |
| ocr_text = "" | |
| print(f"OCR failed for {source} page {page}: {exc}", file=sys.stderr) | |
| sim = similarity(clean_text(manual), clean_text(ocr_text)) | |
| manual_formulas = extract_formula_lines(manual) | |
| ocr_formulas = extract_formula_lines(ocr_text) | |
| results.append({ | |
| "source": source, | |
| "page": page, | |
| "image": str(img), | |
| "similarity": sim, | |
| "manual_chars": len(manual), | |
| "ocr_chars": len(ocr_text), | |
| "manual_formula_lines": manual_formulas, | |
| "ocr_formula_lines": ocr_formulas, | |
| }) | |
| print(f"{source} p{page}: sim={sim:.2f} manual={len(manual)} ocr={len(ocr_text)}") | |
| args.output.parent.mkdir(parents=True, exist_ok=True) | |
| args.output.write_text(json.dumps(results, indent=2, ensure_ascii=False)) | |
| print(f"Wrote {len(results)} comparisons to {args.output}") | |
| return 0 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |