""" Smoke Signal v1 — Gradio Tab Module ===================================== Drop this file into your Codex_Extractor Space root. Then add to app.py: from smoke_signal_tab import smoke_signal_tab, SS_CSS # Add SS_CSS to your existing CSS string # Add smoke_signal_tab() call inside your gr.Blocks() tabs Architecture: Step 1: INGEST+PROFILE — upload PDFs, register, auto-profile Step 2: PROFILE (optional) — manual re-run when needed Step 3: OCR — Surya extraction + confidence scoring Step 4: REVIEW — human correction workbench (feeds training data) Step 5: EXPORT — clean JSONL to Codex + downloadable gold set Self-improvement loop: Every correction → recalibrates confidence thresholds in real time Every correction → appended to gold_corrections.jsonl for fine-tuning """ import csv import concurrent.futures import hashlib import importlib.util import json import os import re import tempfile import threading import time from datetime import datetime from pathlib import Path from typing import Optional import gradio as gr import pandas as pd # ── Paths ────────────────────────────────────────────────────────────────────── # Use /tmp for all data — writable on HF Spaces, persists within a session SS_ROOT = Path(os.environ.get("SS_DATA_ROOT", "/tmp/smoke_signal")) SOURCE_DIR = SS_ROOT / "source_pdfs" MANIFEST_CSV = SS_ROOT / "manifest" / "source_manifest.csv" PROFILES_DIR = SS_ROOT / "manifest" / "page_profiles" OCR_RAW_DIR = SS_ROOT / "ocr_raw" RENDERS_DIR = SS_ROOT / "renders" REGIONS_DIR = SS_ROOT / "regions" REVIEW_DIR = SS_ROOT / "review" EXPORTS_DIR = SS_ROOT / "exports" GOLD_DIR = SS_ROOT / "gold" LOGS_DIR = SS_ROOT / "logs" for d in [SOURCE_DIR, MANIFEST_CSV.parent, PROFILES_DIR, OCR_RAW_DIR, RENDERS_DIR, REGIONS_DIR, REVIEW_DIR, EXPORTS_DIR, GOLD_DIR, LOGS_DIR]: d.mkdir(parents=True, exist_ok=True) GOLD_FILE = GOLD_DIR / "gold_corrections.jsonl" NOISE_DIR = SS_ROOT / "calibration" / "noise_patterns" NOISE_GLOBAL_FILE = NOISE_DIR / "_global_noise.json" NOISE_MIN_CHARS = int(os.environ.get("SS_NOISE_MIN_CHARS", "3")) NOISE_MATCH_MIN_PATTERN_COVERAGE = float(os.environ.get("SS_NOISE_MATCH_MIN_PATTERN_COVERAGE", "0.65")) NOISE_MATCH_MIN_TEXT_COVERAGE = float(os.environ.get("SS_NOISE_MATCH_MIN_TEXT_COVERAGE", "0.08")) _NOISE_IO_LOCK = threading.RLock() def _normalize_noise_text(text: str) -> str: txt = (text or "").lower() txt = re.sub(r"[\r\n\t]+", " ", txt) txt = re.sub(r"[^a-z0-9\u4e00-\u9fff\s]+", " ", txt) txt = re.sub(r"\s+", " ", txt).strip() return txt def _noise_path(book_id: str) -> Path: NOISE_DIR.mkdir(parents=True, exist_ok=True) return NOISE_DIR / f"{book_id}_noise.json" def _load_noise_patterns_for_scope(path: Path) -> list[str]: if not path.exists(): return [] try: data = json.loads(path.read_text()) except Exception: return [] if not isinstance(data, list): return [] out: list[str] = [] for p in data: raw = str(p or "").strip() norm = _normalize_noise_text(raw) if len(norm) < NOISE_MIN_CHARS: continue out.append(raw) return out def _write_json_atomic(path: Path, payload) -> None: path.parent.mkdir(parents=True, exist_ok=True) fd, tmp_name = tempfile.mkstemp(prefix=path.name + ".", suffix=".tmp", dir=str(path.parent)) tmp_path = Path(tmp_name) try: with os.fdopen(fd, "w", encoding="utf-8") as fh: json.dump(payload, fh, indent=2, ensure_ascii=False) fh.flush() os.fsync(fh.fileno()) os.replace(str(tmp_path), str(path)) finally: if tmp_path.exists(): try: tmp_path.unlink() except Exception: pass def _load_noise_patterns(book_id: str, include_global: bool = True) -> list[str]: combined: list[str] = [] seen: set[str] = set() paths = [_noise_path(book_id)] if include_global: paths.append(NOISE_GLOBAL_FILE) with _NOISE_IO_LOCK: for path in paths: for pattern in _load_noise_patterns_for_scope(path): norm = _normalize_noise_text(pattern) if norm in seen: continue seen.add(norm) combined.append(pattern) return combined def _save_noise_pattern(book_id: str, pattern: str, save_global: bool = True) -> bool: pattern = str(pattern or "").strip() norm = _normalize_noise_text(pattern) if len(norm) < NOISE_MIN_CHARS: return False changed = False targets = [_noise_path(book_id)] if save_global: targets.append(NOISE_GLOBAL_FILE) with _NOISE_IO_LOCK: for target in targets: existing = _load_noise_patterns_for_scope(target) existing_norm = {_normalize_noise_text(p) for p in existing} if norm in existing_norm: continue existing.append(pattern) _write_json_atomic(target, existing) changed = True return changed def _json_default(obj): # numpy / pandas scalar safety for json dumps try: import numpy as np # local import to avoid hard dependency at import-time if isinstance(obj, np.generic): return obj.item() except Exception: pass if isinstance(obj, Path): return str(obj) # pandas Timestamp/NA etc. try: if isinstance(obj, pd.Timestamp): return obj.isoformat() except Exception: pass # Generic numeric coercion fallback try: if hasattr(obj, "item"): return obj.item() except Exception: pass return str(obj) def _text_matches_noise(text: str, patterns: list) -> bool: if not patterns or not text: return False normalized_text = _normalize_noise_text(text) if not normalized_text: return False text_tokens = set(normalized_text.split()) if not text_tokens: return False for pattern in patterns: pat_norm = _normalize_noise_text(str(pattern or "")) if not pat_norm: continue # Best signal for recurring watermarks/headers if pat_norm in normalized_text: return True pat_tokens = set(pat_norm.split()) if not pat_tokens: continue overlap_count = len(text_tokens & pat_tokens) if overlap_count == 0: continue pattern_coverage = overlap_count / len(pat_tokens) text_coverage = overlap_count / len(text_tokens) if ( pattern_coverage >= NOISE_MATCH_MIN_PATTERN_COVERAGE and text_coverage >= NOISE_MATCH_MIN_TEXT_COVERAGE ): return True return False DECISIONS_CSV = REVIEW_DIR / "review_decisions.csv" QUEUE_CSV = REVIEW_DIR / "review_queue.csv" BANNER_DATA_URI_FILE = Path(__file__).resolve().parent / "assets" / "smoke_signal_banner_data_uri.txt" PUNCT_CORRECTOR_PATH = Path(__file__).resolve().parent / "smoke_signal" / "scripts" / "punct_corrector.py" FONT_LIBRARY_PATH = Path(__file__).resolve().parent / "smoke_signal" / "scripts" / "font_library.py" MANIFEST_COLUMNS = [ "book_id", "filename", "sha256", "page_count", "rights_class", "status", "acquisition_date", "notes", "story_pages_include", "story_pages_exclude", "safe_title", ] MANIFEST_TEXT_COLUMNS = { "book_id", "filename", "sha256", "rights_class", "status", "acquisition_date", "notes", "story_pages_include", "story_pages_exclude", "safe_title", } def _load_banner_image_css() -> str: """ Return a CSS-ready background-image value. Uses a text data-URI file so HF push is text-only (no binary file tracking needed). """ try: uri = BANNER_DATA_URI_FILE.read_text(encoding="utf-8").strip() if not uri: return "none" if not uri.startswith("data:image/"): return "none" return f"url('{uri}')" except Exception: return "none" SS_BANNER_IMAGE_CSS = _load_banner_image_css() _PUNCT_MODULE = None _PUNCT_MODULE_ERROR = None _FONT_MODULE = None _FONT_MODULE_ERROR = None def _load_punct_module(): global _PUNCT_MODULE, _PUNCT_MODULE_ERROR if _PUNCT_MODULE is not None: return _PUNCT_MODULE if _PUNCT_MODULE_ERROR is not None: return None if not PUNCT_CORRECTOR_PATH.exists(): _PUNCT_MODULE_ERROR = f"not found: {PUNCT_CORRECTOR_PATH}" return None try: spec = importlib.util.spec_from_file_location("smoke_signal_punct_corrector", str(PUNCT_CORRECTOR_PATH)) if spec is None or spec.loader is None: _PUNCT_MODULE_ERROR = "invalid import spec" return None module = importlib.util.module_from_spec(spec) spec.loader.exec_module(module) _PUNCT_MODULE = module return _PUNCT_MODULE except Exception as e: _PUNCT_MODULE_ERROR = str(e) return None def _load_font_module(): global _FONT_MODULE, _FONT_MODULE_ERROR if _FONT_MODULE is not None: return _FONT_MODULE if _FONT_MODULE_ERROR is not None: return None if not FONT_LIBRARY_PATH.exists(): _FONT_MODULE_ERROR = f"not found: {FONT_LIBRARY_PATH}" return None try: spec = importlib.util.spec_from_file_location("smoke_signal_font_library", str(FONT_LIBRARY_PATH)) if spec is None or spec.loader is None: _FONT_MODULE_ERROR = "invalid import spec" return None module = importlib.util.module_from_spec(spec) spec.loader.exec_module(module) _FONT_MODULE = module return _FONT_MODULE except Exception as e: _FONT_MODULE_ERROR = str(e) return None def _apply_punctuation_corrections(text: str, book_id: str, font_name: Optional[str] = None) -> tuple[str, list, float]: module = _load_punct_module() if module is None: return text, [], 1.0 try: corrected, flags, score = module.apply_punctuation_corrections( text or "", book_id=book_id, font_name=font_name, ) return corrected, flags, float(score) except Exception: return text, [], 1.0 def _punctuation_confidence_penalty(text: str, confidence: float) -> float: module = _load_punct_module() if module is None: return float(confidence) try: return float(module.punctuation_confidence_penalty(text or "", float(confidence))) except Exception: return float(confidence) def _record_punctuation_correction( raw_text: str, gold_text: str, book_id: str, font_name: Optional[str] = None, ) -> int: module = _load_punct_module() if module is None: return 0 try: return int( module.record_punctuation_correction( raw_text or "", gold_text or "", book_id=book_id, font_name=font_name, ) ) except Exception: return 0 def _identify_page_font(render_abs_path: str) -> dict: module = _load_font_module() if module is None: return { "font_name": None, "confidence": 0.0, "alternatives": [], "image_url": None, "error": _FONT_MODULE_ERROR or "font module unavailable", } try: result = module.identify_page_font(render_abs_path) return result if isinstance(result, dict) else { "font_name": None, "confidence": 0.0, "alternatives": [], "image_url": None, "error": "invalid response from font module", } except Exception as e: return { "font_name": None, "confidence": 0.0, "alternatives": [], "image_url": None, "error": str(e), } def _resolve_tesseract_lang(font_name: Optional[str]) -> str: module = _load_font_module() if module is None: return "eng" try: lang = str(module.resolve_tesseract_lang(font_name) or "").strip() return lang if lang else "eng" except Exception: return "eng" def _resolve_tessdata_dir(model_name: Optional[str]) -> Optional[str]: module = _load_font_module() if module is None: return None try: value = module.resolve_tessdata_dir(model_name) return str(value) if value else None except Exception: return None def _update_font_engine_stats(font_name: Optional[str], surya_conf: float, tess_conf: float) -> None: module = _load_font_module() if module is None: return try: module.update_font_engine_stats(font_name, surya_conf, tess_conf) except Exception: return def _mixfont_preflight() -> dict: module = _load_font_module() if module is None: return { "api_key_set": False, "api_url": "", "image_url_template_set": False, "image_base_set": False, "space_host_set": False, "public_image_url_source_available": False, "error": _FONT_MODULE_ERROR or "font module unavailable", } try: data = module.mixfont_preflight() if not isinstance(data, dict): return { "api_key_set": False, "api_url": "", "image_url_template_set": False, "image_base_set": False, "space_host_set": False, "public_image_url_source_available": False, "error": "invalid preflight response", } return data except Exception as e: return { "api_key_set": False, "api_url": "", "image_url_template_set": False, "image_base_set": False, "space_host_set": False, "public_image_url_source_available": False, "error": str(e), } def _punctuation_map_summary() -> dict: module = _load_punct_module() if module is None: return {"total_pairs": 0, "top_substitutions": []} try: summary = module.correction_map_summary() if not isinstance(summary, dict): return {"total_pairs": 0, "top_substitutions": []} return summary except Exception: return {"total_pairs": 0, "top_substitutions": []} # ── Confidence calibration state (in-memory, persisted to disk) ──────────────── CALIBRATION_FILE = SS_ROOT / "manifest" / "confidence_calibration.json" DEFAULT_CALIBRATION = { "narration": {"auto_accept": 0.85, "review": 0.60, "quarantine": 0.35, "corrections": 0}, "dialogue-speech-bubble": {"auto_accept": 0.80, "review": 0.55, "quarantine": 0.30, "corrections": 0}, "caption": {"auto_accept": 0.82, "review": 0.58, "quarantine": 0.32, "corrections": 0}, "title": {"auto_accept": 0.88, "review": 0.65, "quarantine": 0.40, "corrections": 0}, "sign-label": {"auto_accept": 0.75, "review": 0.50, "quarantine": 0.25, "corrections": 0}, "_default": {"auto_accept": 0.85, "review": 0.60, "quarantine": 0.35, "corrections": 0}, } def load_calibration() -> dict: if CALIBRATION_FILE.exists(): try: return json.load(open(CALIBRATION_FILE)) except Exception: pass return DEFAULT_CALIBRATION.copy() def save_calibration(cal: dict) -> None: with open(CALIBRATION_FILE, "w") as f: json.dump(cal, f, indent=2) def recalibrate(region_class: str, was_correct: bool, confidence: float) -> None: """Tighten thresholds when corrections happen frequently for a region class.""" cal = load_calibration() key = region_class if region_class in cal else "_default" entry = cal[key] if not was_correct: entry["corrections"] = entry.get("corrections", 0) + 1 corrections = entry["corrections"] # Every 5 corrections on same class: tighten auto-accept by 2% if corrections % 5 == 0: entry["auto_accept"] = min(0.98, entry["auto_accept"] + 0.02) entry["review"] = min(0.90, entry["review"] + 0.01) cal[key] = entry save_calibration(cal) # ── CSS ──────────────────────────────────────────────────────────────────────── SS_CSS = """ /* ── Smoke Signal palette ── */ :root { --ss-bg: #0d1117; --ss-surface: #161b22; --ss-border: #30363d; --ss-smoke: #8b949e; --ss-signal: #f0883e; --ss-glow: #58a6ff; --ss-green: #3fb950; --ss-red: #f85149; --ss-gold: #e3b341; --ss-text: #e6edf3; --ss-muted: #7d8590; } /* Hero banner + wizard */ .ss-hero { position: relative; overflow: hidden; border-radius: 16px; border: 1px solid #2a3240; margin: 12px 0 16px; padding-top: 230px; background: linear-gradient(132deg, #040816 0%, #0a1329 45%, #030916 100%); box-shadow: 0 24px 70px rgba(3, 9, 25, 0.38); } .ss-hero::before { content: ""; position: absolute; inset: 0; background-image: radial-gradient(circle at 12% 18%, rgba(90, 164, 255, 0.35), transparent 25%); opacity: 0.7; pointer-events: none; } .ss-hero-art { position: absolute; top: 10px; left: 10px; right: 10px; height: 205px; border-radius: 14px; border: 1px solid rgba(125, 133, 144, 0.24); background-image: __SS_BANNER_IMAGE_CSS__; background-size: cover; background-position: center top; box-shadow: inset 0 -26px 40px rgba(3, 9, 22, 0.65); z-index: 1; } .ss-hero-step-wrap { position: relative; z-index: 2; margin: 0 20px 20px; border-radius: 14px; border: 1px solid rgba(125, 133, 144, 0.26); background: linear-gradient(180deg, rgba(11, 18, 35, 0.88), rgba(8, 14, 30, 0.88)); padding: 12px 10px 10px; } .ss-wizard { display: flex; align-items: center; gap: 0; overflow-x: auto; padding: 2px 4px; } .ss-step { position: relative; display: inline-flex; align-items: center; gap: 10px; padding: 12px 14px 18px; cursor: default; transition: all 0.2s; white-space: nowrap; font-family: 'Source Code Pro', 'Courier New', monospace; font-size: 11px; font-weight: 700; color: #7f89a0; letter-spacing: 1px; text-transform: uppercase; } .ss-step.active { color: #ff8e56; } .ss-step.active::after { content: ""; position: absolute; left: 6px; right: 6px; bottom: 2px; height: 3px; border-radius: 99px; background: linear-gradient(90deg, #ff6c4a, #ffb357); box-shadow: 0 0 14px rgba(240, 136, 62, 0.42); } .ss-num { width: 26px; height: 26px; border-radius: 999px; display: flex; align-items: center; justify-content: center; font-size: 11px; font-weight: 900; background: rgba(11, 18, 35, 0.74); border: 2px solid currentColor; flex-shrink: 0; } .ss-connector { width: 44px; height: 2px; background: linear-gradient(90deg, rgba(104, 132, 180, 0.4), rgba(104, 132, 180, 0.15)); flex-shrink: 0; border-radius: 999px; } @media (max-width: 900px) { .ss-hero { padding-top: 156px; } .ss-hero-art { height: 134px; } } /* Main panel */ .ss-panel { background: var(--ss-bg); min-height: 600px; padding: 28px; font-family: 'Lato', sans-serif; color: var(--ss-text); } .ss-panel-header { display: flex; align-items: center; gap: 16px; margin-bottom: 28px; padding-bottom: 20px; border-bottom: 1px solid var(--ss-border); } .ss-panel-icon { font-size: 32px; width: 56px; height: 56px; display: flex; align-items: center; justify-content: center; background: var(--ss-surface); border: 1px solid var(--ss-border); border-radius: 10px; } .ss-panel-title { font-family: 'Playfair Display', Georgia, serif; font-size: 22px; font-weight: 700; color: var(--ss-text); margin: 0; } .ss-panel-sub { font-family: 'Source Code Pro', monospace; font-size: 11px; color: var(--ss-muted); letter-spacing: 2px; text-transform: uppercase; margin: 4px 0 0; } /* Status pills */ .ss-pill { display: inline-flex; align-items: center; gap: 6px; padding: 4px 12px; border-radius: 999px; font-size: 11px; font-weight: 700; font-family: 'Source Code Pro', monospace; letter-spacing: 1px; text-transform: uppercase; } .ss-pill-waiting { background: #21262d; color: var(--ss-muted); border: 1px solid var(--ss-border); } .ss-pill-running { background: #1c2a1e; color: var(--ss-gold); border: 1px solid var(--ss-gold); } .ss-pill-done { background: #1a2f1a; color: var(--ss-green); border: 1px solid var(--ss-green); } .ss-pill-error { background: #2d1a1a; color: var(--ss-red); border: 1px solid var(--ss-red); } .ss-pill-review { background: #2a1f0e; color: var(--ss-signal); border: 1px solid var(--ss-signal); } /* Cards */ .ss-card { background: var(--ss-surface); border: 1px solid var(--ss-border); border-radius: 10px; padding: 20px; margin-bottom: 16px; } .ss-card-title { font-size: 13px; font-weight: 700; color: var(--ss-smoke); text-transform: uppercase; letter-spacing: 2px; margin-bottom: 12px; font-family: 'Source Code Pro', monospace; } /* Metric row */ .ss-metrics { display: grid; grid-template-columns: repeat(4, 1fr); gap: 12px; margin-bottom: 20px; } .ss-metric { background: var(--ss-surface); border: 1px solid var(--ss-border); border-radius: 8px; padding: 16px; text-align: center; } .ss-metric-val { font-size: 28px; font-weight: 900; font-family: 'Source Code Pro', monospace; color: var(--ss-text); line-height: 1; } .ss-metric-label { font-size: 10px; color: var(--ss-muted); text-transform: uppercase; letter-spacing: 2px; margin-top: 6px; } /* Progress bar */ .ss-progress-wrap { background: var(--ss-border); border-radius: 4px; height: 6px; margin: 8px 0; overflow: hidden; } .ss-progress-bar { height: 6px; border-radius: 4px; background: linear-gradient(90deg, var(--ss-signal), var(--ss-gold)); transition: width 0.4s ease; } /* Review workbench */ .ss-review-grid { display: grid; grid-template-columns: 240px 1fr 320px; gap: 16px; height: 580px; } .ss-queue-list { background: var(--ss-surface); border: 1px solid var(--ss-border); border-radius: 8px; overflow-y: auto; padding: 8px; } .ss-queue-item { padding: 10px 12px; border-radius: 6px; margin-bottom: 6px; cursor: pointer; border-left: 3px solid var(--ss-border); font-size: 12px; transition: all 0.15s; } .ss-queue-item:hover { background: #21262d; } .ss-queue-item.active { background: #1c2028; border-left-color: var(--ss-signal); } .ss-queue-item.done { border-left-color: var(--ss-green); opacity: 0.7; } .ss-queue-item.quar { border-left-color: var(--ss-red); } .ss-image-panel { background: #010409; border: 1px solid var(--ss-border); border-radius: 8px; display: flex; align-items: center; justify-content: center; overflow: hidden; } .ss-action-panel { background: var(--ss-surface); border: 1px solid var(--ss-border); border-radius: 8px; padding: 16px; display: flex; flex-direction: column; gap: 12px; overflow-y: auto; } /* Buttons */ .ss-btn-accept { background: var(--ss-green) !important; color: #010409 !important; font-weight: 800 !important; border-radius: 6px !important; } .ss-btn-edit { background: var(--ss-signal) !important; color: #010409 !important; font-weight: 800 !important; border-radius: 6px !important; } .ss-btn-reject { background: var(--ss-red) !important; color: white !important; font-weight: 800 !important; border-radius: 6px !important; } .ss-btn-quar { background: #21262d !important; color: var(--ss-gold) !important; font-weight: 800 !important; border-radius: 6px !important; border: 1px solid var(--ss-gold) !important; } .ss-btn-next { background: var(--ss-glow) !important; color: #010409 !important; font-weight: 800 !important; border-radius: 6px !important; } .ss-btn-run { background: linear-gradient(135deg, var(--ss-signal), var(--ss-gold)) !important; color: #010409 !important; font-weight: 900 !important; border-radius: 8px !important; font-size: 15px !important; min-height: 52px !important; } /* Training signal */ .ss-training-badge { display: inline-flex; align-items: center; gap: 6px; padding: 6px 12px; background: #1a2535; border: 1px solid var(--ss-glow); border-radius: 6px; font-size: 11px; color: var(--ss-glow); font-family: 'Source Code Pro', monospace; } .ss-pulse { width: 8px; height: 8px; border-radius: 50%; background: var(--ss-glow); animation: ss-pulse 1.5s infinite; } @keyframes ss-pulse { 0%, 100% { opacity: 1; transform: scale(1); } 50% { opacity: 0.4; transform: scale(0.8); } } /* Log terminal */ .ss-log { background: #010409; border: 1px solid var(--ss-border); border-radius: 8px; padding: 14px; font-family: 'Source Code Pro', 'Courier New', monospace; font-size: 12px; color: #7ee787; min-height: 120px; max-height: 200px; overflow-y: auto; white-space: pre-wrap; } /* Gradio overrides for dark theme inside SS */ #ss-tab .gradio-container { background: var(--ss-bg) !important; } #ss-tab textarea, #ss-tab input[type=text] { background: var(--ss-surface) !important; border: 1px solid var(--ss-border) !important; color: var(--ss-text) !important; border-radius: 6px !important; font-family: 'Source Code Pro', monospace !important; font-size: 13px !important; } #ss-tab .label-wrap span { color: var(--ss-smoke) !important; font-size: 11px !important; text-transform: uppercase !important; letter-spacing: 1px !important; } """ SS_CSS = SS_CSS.replace("__SS_BANNER_IMAGE_CSS__", SS_BANNER_IMAGE_CSS) # ── Utility functions ────────────────────────────────────────────────────────── def sha256_file(path: Path) -> str: h = hashlib.sha256() with open(path, "rb") as f: for block in iter(lambda: f.read(1 << 20), b""): h.update(block) return h.hexdigest() def load_manifest_df() -> pd.DataFrame: if not MANIFEST_CSV.exists(): return pd.DataFrame(columns=MANIFEST_COLUMNS) df = pd.read_csv(MANIFEST_CSV, dtype=str, keep_default_na=False) for col in MANIFEST_COLUMNS: if col not in df.columns: df[col] = "" for col in MANIFEST_TEXT_COLUMNS: if col in df.columns: df[col] = df[col].fillna("").astype(str) if "page_count" in df.columns: df["page_count"] = df["page_count"].fillna("").astype(str) return df[MANIFEST_COLUMNS] def save_manifest_df(df: pd.DataFrame) -> None: out = df.copy() for col in MANIFEST_COLUMNS: if col not in out.columns: out[col] = "" for col in MANIFEST_TEXT_COLUMNS: if col in out.columns: out[col] = out[col].fillna("").astype(str) if "page_count" in out.columns: out["page_count"] = out["page_count"].fillna("").astype(str) out[MANIFEST_COLUMNS].to_csv(MANIFEST_CSV, index=False) def next_book_id(df: pd.DataFrame) -> str: existing = set(df["book_id"].tolist()) if not df.empty else set() for i in range(1, 10000): bid = f"SS-BOOK-{i:04d}" if bid not in existing: return bid return "SS-BOOK-9999" _TITLE_STOPWORDS = {"the", "a", "an"} def _derive_book_code(title: str, code_hint: str = "") -> str: hint = re.sub(r"[^a-z]", "", str(code_hint or "").lower()) if len(hint) >= 3: return hint[:3] words = re.findall(r"[a-z]+", str(title or "").lower()) if words and words[0] in _TITLE_STOPWORDS and len(words) > 1: words = words[1:] letters = "".join(words) if not letters: return "bok" consonants = "".join(ch for ch in letters if ch not in "aeiou") base = consonants[:3] if len(consonants) >= 3 else letters[:3] return (base + "xxx")[:3] def _extract_year(*parts: str) -> str: for part in parts: match = re.search(r"\b(1[6-9]\d{2}|20\d{2})\b", str(part or "")) if match: return match.group(1) return "" def _suggest_book_id( title: str, notes: str = "", code_hint: str = "", year_hint: str = "", existing_ids: Optional[set[str]] = None, current_id: str = "", ) -> str: code = _derive_book_code(title, code_hint=code_hint) year = _extract_year(year_hint, notes, title) or "0000" base = f"{code}{year}" if existing_ids is None: return base if base not in existing_ids or base == current_id: return base for suffix in "abcdefghijklmnopqrstuvwxyz": candidate = f"{base}{suffix}" if candidate not in existing_ids or candidate == current_id: return candidate return base def _rename_book_id_references(old_id: str, new_id: str) -> None: if not old_id or not new_id or old_id == new_id: return # Profile JSON old_profile = PROFILES_DIR / f"{old_id}_page_profile.json" new_profile = PROFILES_DIR / f"{new_id}_page_profile.json" if old_profile.exists(): try: data = json.load(open(old_profile, encoding="utf-8")) data["book_id"] = new_id with open(new_profile, "w", encoding="utf-8") as f: json.dump(data, f, indent=2) old_profile.unlink(missing_ok=True) except Exception: pass # Render directory old_render_dir = RENDERS_DIR / old_id new_render_dir = RENDERS_DIR / new_id if old_render_dir.exists() and not new_render_dir.exists(): old_render_dir.rename(new_render_dir) # OCR raw directory + file old_ocr_dir = OCR_RAW_DIR / old_id new_ocr_dir = OCR_RAW_DIR / new_id if old_ocr_dir.exists() and not new_ocr_dir.exists(): old_ocr_dir.rename(new_ocr_dir) if new_ocr_dir.exists(): old_raw = new_ocr_dir / f"{old_id}_ocr_raw.json" new_raw = new_ocr_dir / f"{new_id}_ocr_raw.json" if old_raw.exists() and not new_raw.exists(): old_raw.rename(new_raw) if new_raw.exists(): try: raw_data = json.load(open(new_raw, encoding="utf-8")) raw_data["book_id"] = new_id with open(new_raw, "w", encoding="utf-8") as f: json.dump(raw_data, f, indent=2) except Exception: pass # Queue and decision CSVs for csv_path in [QUEUE_CSV, DECISIONS_CSV]: if not csv_path.exists(): continue try: cdf = pd.read_csv(csv_path) if "book_id" in cdf.columns: cdf.loc[cdf["book_id"] == old_id, "book_id"] = new_id if "region_id" in cdf.columns: region_series = cdf["region_id"].astype(str) mask = region_series.str.startswith(f"{old_id}_") cdf.loc[mask, "region_id"] = region_series[mask].str.replace( f"{old_id}_", f"{new_id}_", n=1, regex=False ) cdf.to_csv(csv_path, index=False) except Exception: pass # Gold JSONL if GOLD_FILE.exists(): tmp_path = GOLD_FILE.with_suffix(".tmp") try: with open(GOLD_FILE, "r", encoding="utf-8") as src, open(tmp_path, "w", encoding="utf-8") as dst: for line in src: line = line.strip() if not line: continue try: obj = json.loads(line) except Exception: dst.write(line + "\n") continue if obj.get("book_id") == old_id: obj["book_id"] = new_id region_id = str(obj.get("region_id", "")) if region_id.startswith(f"{old_id}_"): obj["region_id"] = region_id.replace(f"{old_id}_", f"{new_id}_", 1) dst.write(json.dumps(obj, ensure_ascii=False) + "\n") tmp_path.replace(GOLD_FILE) except Exception: if tmp_path.exists(): tmp_path.unlink(missing_ok=True) def load_queue_df() -> pd.DataFrame: if not QUEUE_CSV.exists(): return pd.DataFrame() return pd.read_csv(QUEUE_CSV) def load_decisions_df() -> pd.DataFrame: if not DECISIONS_CSV.exists(): return pd.DataFrame() return pd.read_csv(DECISIONS_CSV) def ts() -> str: return datetime.utcnow().strftime("%Y%m%d-%H%M%S") def log_line(msg: str) -> str: return f"[{datetime.utcnow().strftime('%H:%M:%S')}] {msg}" def _parse_page_selection(spec: str, max_page: int | None = None) -> tuple[set[int] | None, str | None]: """ Parse optional page-selection text. Accepted forms: - empty / all / * -> None (means all pages) - "7" - "3-8" - "1,3,5-7" """ raw = (spec or "").strip().lower() if raw in ("", "all", "*"): return None, None out: set[int] = set() for token in [t.strip() for t in raw.split(",") if t.strip()]: if "-" in token: parts = token.split("-", 1) if len(parts) != 2 or (not parts[0].isdigit()) or (not parts[1].isdigit()): return None, f"Invalid page range token: '{token}'" start = int(parts[0]) end = int(parts[1]) if start <= 0 or end <= 0: return None, f"Pages must be >= 1 (token: '{token}')" if end < start: return None, f"Range end before start (token: '{token}')" out.update(range(start, end + 1)) else: if not token.isdigit(): return None, f"Invalid page token: '{token}'" page = int(token) if page <= 0: return None, f"Pages must be >= 1 (token: '{token}')" out.add(page) if max_page is not None: out = {p for p in out if p <= int(max_page)} if not out: return None, f"No selected pages fall within this PDF (max page {max_page})." return out, None def _safe_title_slug(title: str) -> str: raw = str(title or "").strip().lower() if not raw: return "" slug = re.sub(r"[^a-z0-9]+", "-", raw).strip("-") return slug[:120] def _resolve_book_row(df: pd.DataFrame, selector: str) -> Optional[pd.Series]: if df.empty: return None raw = str(selector or "").strip() if not raw: return df.iloc[-1] # Exact book_id exact = df[df["book_id"].astype(str) == raw] if not exact.empty: return exact.iloc[0] low = raw.lower() # Case-insensitive book_id bid_match = df[df["book_id"].astype(str).str.lower() == low] if not bid_match.empty: return bid_match.iloc[0] # Safe title match if "safe_title" in df.columns: st_match = df[df["safe_title"].astype(str).str.lower() == low] if not st_match.empty: return st_match.iloc[0] # Filename / stem match name_match = df[df["filename"].astype(str).str.lower() == low] if not name_match.empty: return name_match.iloc[0] stem_match = df[df["filename"].astype(str).str.lower().str.replace(".pdf", "", regex=False) == low] if not stem_match.empty: return stem_match.iloc[0] return None def _resolve_uploaded_file_path(file_obj) -> tuple[Optional[Path], str]: """ Robustly resolve an uploaded file path across Gradio runtime object shapes. Returns (path, debug_hint). """ candidates: list[str] = [] if file_obj is None: return None, "upload item is None" # Candidate 1: direct string form (often full temp path in Gradio) try: s = str(file_obj).strip() if s: candidates.append(s) except Exception: pass # Candidate 2: .name attribute (file-like wrappers) try: n = getattr(file_obj, "name", "") n = str(n).strip() if n: candidates.append(n) except Exception: pass # Candidate 3: explicit path attr used by some wrappers try: p = getattr(file_obj, "path", "") p = str(p).strip() if p: candidates.append(p) except Exception: pass # Deduplicate in order seen = set() uniq = [] for c in candidates: if c not in seen: uniq.append(c) seen.add(c) for c in uniq: try: path = Path(c) if path.exists(): return path, f"resolved from '{c}'" except Exception: continue return None, f"no existing path in candidates={uniq!r}" def _clean_page_spec(spec: str) -> str: raw = (spec or "").strip().lower() if raw in ("", "all", "*", "none", "-"): return "" parsed, err = _parse_page_selection(raw) if err: raise ValueError(err) if parsed is None: return "" return ",".join(str(p) for p in sorted(parsed)) def _normalize_saved_spec(value) -> str: raw = str(value if value is not None else "").strip().lower() if raw in ("", "nan", "none", "null", "-", "all", "*"): return "" return raw def _default_ocr_scope_values() -> tuple[str, str]: """ Prefill OCR page selectors from saved ingest scope. If multiple books exist, use the latest manifest row as default. """ df = load_manifest_df() if df.empty: return "", "" row = df.iloc[-1] include_val = _normalize_saved_spec(row.get("story_pages_include", "")) exclude_val = _normalize_saved_spec(row.get("story_pages_exclude", "")) return include_val, exclude_val def save_book_scope(book_id: str, include_spec: str, exclude_spec: str, safe_title: str) -> tuple: df = load_manifest_df() if df.empty: return _ingest_status_html("idle"), df, "No books in manifest yet." row = _resolve_book_row(df, book_id) if row is None: return _ingest_status_html("idle"), df, f"Book ID or title '{book_id}' not found." bid = str(row["book_id"]) try: include_clean = _clean_page_spec(include_spec) except ValueError as e: return _ingest_status_html("idle"), df, f"Invalid include pages: {e}" try: exclude_clean = _clean_page_spec(exclude_spec) except ValueError as e: return _ingest_status_html("idle"), df, f"Invalid exclude pages: {e}" title_raw = (safe_title or "").strip() if title_raw: safe = _safe_title_slug(title_raw) else: filename = str(row.get("filename", "") or "") stem = Path(filename).stem if filename else bid safe = _safe_title_slug(stem) df.loc[df["book_id"] == bid, "story_pages_include"] = include_clean df.loc[df["book_id"] == bid, "story_pages_exclude"] = exclude_clean df.loc[df["book_id"] == bid, "safe_title"] = safe save_manifest_df(df) line1 = log_line(f"✓ Saved scope for {bid}") line2 = log_line( f"ℹ include={include_clean or 'all'} · exclude={exclude_clean or 'none'} · safe_title={safe}" ) msg = f"{line1}\n{line2}" return _ingest_status_html("done"), load_manifest_df(), msg # ── Step 1: INGEST ───────────────────────────────────────────────────────────── def ingest_pdfs(files, rights_class: str, notes: str, book_code_hint: str = "", publication_year: str = "") -> tuple: """Register uploaded PDFs into the source manifest.""" if not files: return _ingest_status_html("idle"), pd.DataFrame(), "No files uploaded." df = load_manifest_df() log = [] new_count = 0 dup_count = 0 auto_profile_ids = [] for file in files: path, resolve_hint = _resolve_uploaded_file_path(file) if path is None: log.append(log_line(f"⚠ Upload path unresolved: {resolve_hint}")) continue file_hash = sha256_file(path) desired_book_id = _suggest_book_id( Path(path.name).stem, notes=notes, code_hint=book_code_hint, year_hint=publication_year, existing_ids=set(df["book_id"].tolist()) if not df.empty else set(), ) # Check duplicate — update rights/notes if changed if not df.empty and file_hash in df["sha256"].values: existing_book_id = df.loc[df["sha256"] == file_hash, "book_id"].values[0] existing_status = df.loc[df["sha256"] == file_hash, "status"].values[0] existing_rights = df.loc[df["sha256"] == file_hash, "rights_class"].values[0] if False: # disabled: do not rename on duplicate re-registration pass if rights_class != "unknown" and existing_rights != rights_class: df.loc[df["sha256"] == file_hash, "rights_class"] = rights_class df.loc[df["sha256"] == file_hash, "notes"] = notes log.append(log_line(f"↻ Updated rights for duplicate: {path.name} → {rights_class}")) if existing_status == "pending": auto_profile_ids.append(existing_book_id) else: log.append(log_line(f"↩ Duplicate: {path.name} ({existing_book_id}, rights={existing_rights})")) dup_count += 1 continue # Copy to source_pdfs dest = SOURCE_DIR / path.name import shutil try: if path.resolve() != dest.resolve(): shutil.copy2(path, dest) except Exception as e: log.append(log_line(f"✗ Failed to copy {path.name}: {e}")) continue # Page count page_count = None try: import fitz doc = fitz.open(str(dest)) page_count = doc.page_count doc.close() except Exception: pass book_id = desired_book_id if (not book_id) or (book_id in set(df["book_id"].tolist())): book_id = next_book_id(df) new_row = pd.DataFrame([{ "book_id": book_id, "filename": path.name, "sha256": file_hash, "page_count": str(page_count) if page_count else "", "rights_class": rights_class, "status": "pending", "acquisition_date": datetime.utcnow().date().isoformat(), "notes": notes, "story_pages_include": "", "story_pages_exclude": "", "safe_title": _safe_title_slug(Path(path.name).stem), }]) df = pd.concat([df, new_row], ignore_index=True) log.append(log_line(f"✓ Registered {book_id} — {path.name} ({page_count or '?'} pages) [{resolve_hint}]")) new_count += 1 auto_profile_ids.append(book_id) save_manifest_df(df) summary = f"Registered {new_count} new | {dup_count} duplicates skipped" log.append(log_line(summary)) unique_profile_ids = sorted(set(auto_profile_ids)) if unique_profile_ids: log.append(log_line(f"↻ Auto-profile queued for {len(unique_profile_ids)} source(s)")) _, profile_log = run_profile(book_ids=unique_profile_ids) for line in str(profile_log).splitlines(): if line.strip(): log.append(line) df = load_manifest_df() return _ingest_status_html("done", new_count, dup_count), df, "\n".join(log) def _ingest_status_html(state: str, new=0, dups=0) -> str: df = load_manifest_df() total = len(df) pending = len(df[df["status"] == "pending"]) if not df.empty else 0 unknown = len(df[df["rights_class"] == "unknown"]) if not df.empty else 0 alert = "" if unknown > 0: alert = f'
⚠ {unknown} sources have unknown rights class — set before extraction
' return f"""
{total}
Total Sources
{pending}
Pending
{total - pending}
Processed
{unknown}
Unknown Rights
{alert}""" # ── Rights class updater ────────────────────────────────────────────────────── def update_rights(book_id: str, new_rights: str) -> tuple: """Update rights class for an existing book.""" df = load_manifest_df() if df.empty: return _ingest_status_html("idle"), df, "No books in manifest yet." row = _resolve_book_row(df, book_id) if row is None: return _ingest_status_html("idle"), df, f"Book ID or title '{book_id}' not found." resolved_book_id = str(row["book_id"]) prev_status = df.loc[df["book_id"] == resolved_book_id, "status"].values[0] df.loc[df["book_id"] == resolved_book_id, "rights_class"] = new_rights save_manifest_df(df) msg = f"[{datetime.utcnow().strftime('%H:%M:%S')}] ✓ Updated {resolved_book_id} rights → {new_rights}" if prev_status == "pending" and new_rights not in ("unknown", "excluded"): _, profile_log = run_profile(book_ids=[resolved_book_id]) if profile_log: msg = msg + "\n" + str(profile_log) return _ingest_status_html("done"), load_manifest_df(), msg # ── Step 2: PROFILE ──────────────────────────────────────────────────────────── def run_profile(book_ids: Optional[list[str]] = None) -> tuple: """Profile pending PDFs. If book_ids are provided, profile only those sources.""" df = load_manifest_df() if df.empty: return _profile_status_html(), "No sources registered. Complete Step 1 first." pending = df[df["status"] == "pending"] if book_ids: pending = pending[pending["book_id"].isin(book_ids)] if pending.empty: if book_ids: return _profile_status_html(), "No pending PDFs to profile for selected sources." return _profile_status_html(), "No pending PDFs to profile." log = [] try: import fitz except ImportError: return _profile_status_html(), "PyMuPDF not installed. Run: pip install pymupdf" skipped_rights = 0 for _, row in pending.iterrows(): book_id = row["book_id"] filename = row["filename"] pdf_path = SOURCE_DIR / filename if not pdf_path.exists(): log.append(log_line(f"✗ {book_id}: file not found")) continue if row.get("rights_class") in ("unknown", "excluded"): log.append(log_line(f"↩ {book_id}: skipped — rights={row['rights_class']}")) skipped_rights += 1 continue try: doc = fitz.open(str(pdf_path)) pages = [] routes = {"embedded_text": 0, "ocr": 0, "hybrid": 0} for i in range(doc.page_count): page = doc[i] text = page.get_text("text").strip() has_text = len(text) >= 20 has_imgs = len(page.get_images(full=True)) > 0 route = "embedded_text" if has_text and not has_imgs else \ "hybrid" if has_text and has_imgs else "ocr" routes[route] += 1 pages.append({ "page_number": i + 1, "route": route, "char_count": len(text), "has_images": has_imgs, "width_pt": round(page.rect.width, 1), "height_pt": round(page.rect.height, 1), "rotation_deg": page.rotation, "is_spread": (page.rect.width / max(page.rect.height, 1)) >= 1.6, "warnings": [], "render_path": None, "render_dpi": None, }) doc.close() profile = { "book_id": book_id, "filename": filename, "source_hash": row["sha256"], "page_count": len(pages), "config_version": "ss_profiler_v0.1", "profiled_at": datetime.utcnow().isoformat() + "Z", "route_summary": routes, "render_errors": [], "pages": pages, } profile_path = PROFILES_DIR / f"{book_id}_page_profile.json" with open(profile_path, "w") as f: json.dump(profile, f, indent=2) df.loc[df["book_id"] == book_id, "status"] = "profiled" df.loc[df["book_id"] == book_id, "page_count"] = str(len(pages)) log.append(log_line(f"✓ {book_id}: {len(pages)}pp — embed={routes['embedded_text']} ocr={routes['ocr']} hybrid={routes['hybrid']}")) except Exception as e: log.append(log_line(f"✗ {book_id}: {e}")) if skipped_rights: log.append( log_line( f"⚠ {skipped_rights} source(s) skipped due rights_class=unknown/excluded. " "Set rights in Step 1 to auto-profile on update, or rerun Profile manually." ) ) save_manifest_df(df) return _profile_status_html(), "\n".join(log) def _profile_status_html() -> str: df = load_manifest_df() profiled = len(df[df["status"].isin(["profiled","rendered","ocred","reviewed","exported"])]) if not df.empty else 0 total = len(df) pct = int(profiled / max(total, 1) * 100) # Aggregate route stats from all profiles embed = ocr = hybrid = 0 for p in PROFILES_DIR.glob("*_page_profile.json"): try: data = json.load(open(p)) rs = data.get("route_summary", {}) embed += rs.get("embedded_text", 0) ocr += rs.get("ocr", 0) hybrid += rs.get("hybrid", 0) except Exception: pass return f"""
{profiled}/{total}
Profiled
{embed}
Embedded Text
{ocr}
→ OCR
{hybrid}
Hybrid
{pct}% profiled
""" # ── Step 3: OCR ──────────────────────────────────────────────────────────────── _SURYA_RUNTIME = None _SURYA_LOAD_THREAD = None _SURYA_LOAD_ERROR = None _SURYA_LOAD_LOCK = threading.Lock() try: SS_SURYA_BATCH_SIZE = max(1, int(os.environ.get("SS_SURYA_BATCH_SIZE", "4"))) except Exception: SS_SURYA_BATCH_SIZE = 4 try: SS_SURYA_LOAD_TIMEOUT_SEC = max(1, int(os.environ.get("SS_SURYA_LOAD_TIMEOUT_SEC", "20"))) except Exception: SS_SURYA_LOAD_TIMEOUT_SEC = 20 try: SS_RENDER_DPI_SURYA = max(72, int(os.environ.get("SS_RENDER_DPI_SURYA", "300"))) except Exception: SS_RENDER_DPI_SURYA = 300 try: SS_RENDER_DPI_FALLBACK = max(72, int(os.environ.get("SS_RENDER_DPI_FALLBACK", "180"))) except Exception: SS_RENDER_DPI_FALLBACK = 180 try: SS_TESSERACT_WORKERS = max(1, int(os.environ.get("SS_TESSERACT_WORKERS", "3"))) except Exception: SS_TESSERACT_WORKERS = 3 try: SS_TESSERACT_TIMEOUT_SEC = max(1, int(os.environ.get("SS_TESSERACT_TIMEOUT_SEC", "10"))) except Exception: SS_TESSERACT_TIMEOUT_SEC = 10 try: SS_TESSERACT_PSM = max(1, int(os.environ.get("SS_TESSERACT_PSM", "11"))) except Exception: SS_TESSERACT_PSM = 11 try: SS_TESSERACT_MIN_CONF_KEEP = max(0.0, min(1.0, float(os.environ.get("SS_TESSERACT_MIN_CONF_KEEP", "0.58")))) except Exception: SS_TESSERACT_MIN_CONF_KEEP = 0.58 try: SS_TESSERACT_BATCH_HARD_TIMEOUT_SEC = max(15, int(os.environ.get("SS_TESSERACT_BATCH_HARD_TIMEOUT_SEC", "45"))) except Exception: SS_TESSERACT_BATCH_HARD_TIMEOUT_SEC = 45 try: SS_MIXFONT_MAX_DETECT_PAGES = max(1, int(os.environ.get("SS_MIXFONT_MAX_DETECT_PAGES", "2"))) except Exception: SS_MIXFONT_MAX_DETECT_PAGES = 2 try: SS_MIXFONT_MAX_ERRORS = max(1, int(os.environ.get("SS_MIXFONT_MAX_ERRORS", "1"))) except Exception: SS_MIXFONT_MAX_ERRORS = 1 def _load_surya_runtime(): """Load Surya once per app process and reuse between OCR runs.""" global _SURYA_RUNTIME if _SURYA_RUNTIME is not None: return _SURYA_RUNTIME, None, True surya = None surya_error = None # New API (surya-ocr>=0.17 style) try: from surya.foundation import FoundationPredictor from surya.detection import DetectionPredictor from surya.recognition import RecognitionPredictor try: from surya.common.surya.schema import TaskNames task_name = TaskNames.ocr_with_boxes except Exception: task_name = "ocr_with_boxes" foundation_predictor = FoundationPredictor() det_predictor = DetectionPredictor() rec_predictor = RecognitionPredictor(foundation_predictor) surya = { "api": "predictor-v2", "task_name": task_name, "det_predictor": det_predictor, "rec_predictor": rec_predictor, } except Exception as e: surya_error = e # Legacy API (surya-ocr<=0.6 style) if surya is None: try: from surya.ocr import run_ocr as surya_run from surya.model.detection.model import load_model as load_det from surya.model.detection.processor import load_processor as load_det_proc from surya.model.recognition.model import load_model as load_rec from surya.model.recognition.processor import load_processor as load_rec_proc surya = { "api": "legacy-v1", "run": surya_run, "det_model": load_det(), "det_proc": load_det_proc(), "rec_model": load_rec(), "rec_proc": load_rec_proc(), } except Exception as legacy_error: surya_error = f"{surya_error}; legacy={legacy_error}" if surya is not None: _SURYA_RUNTIME = surya return surya, None, False return None, str(surya_error), False def _surya_loader_worker(): """Background loader to avoid blocking OCR forever on slow model downloads.""" global _SURYA_LOAD_ERROR surya, err, _ = _load_surya_runtime() if surya is None: _SURYA_LOAD_ERROR = err else: _SURYA_LOAD_ERROR = None def _get_surya_runtime_with_timeout(timeout_sec: int): """ Return Surya runtime quickly. If model load is still in progress after timeout, caller should fallback this run. """ global _SURYA_LOAD_THREAD if _SURYA_RUNTIME is not None: return _SURYA_RUNTIME, None, True with _SURYA_LOAD_LOCK: if _SURYA_RUNTIME is not None: return _SURYA_RUNTIME, None, True if _SURYA_LOAD_THREAD is None or not _SURYA_LOAD_THREAD.is_alive(): _SURYA_LOAD_THREAD = threading.Thread(target=_surya_loader_worker, daemon=True) _SURYA_LOAD_THREAD.start() loader_thread = _SURYA_LOAD_THREAD loader_thread.join(timeout=timeout_sec) if _SURYA_RUNTIME is not None: return _SURYA_RUNTIME, None, False if loader_thread.is_alive(): return None, f"Surya load exceeded {timeout_sec}s (still loading in background)", False return None, _SURYA_LOAD_ERROR or "Surya load failed", False def _run_surya_batch(images, surya: dict): """Run a batch of PIL images through Surya using either API shape.""" if surya.get("api") == "predictor-v2": return surya["rec_predictor"]( images, task_names=[surya["task_name"]] * len(images), det_predictor=surya["det_predictor"], highres_images=images, math_mode=True, ) return surya["run"]( images, [["en"]] * len(images), surya["det_model"], surya["det_proc"], surya["rec_model"], surya["rec_proc"], ) def _regions_from_page_result(page_result): regions = [] for line in getattr(page_result, "text_lines", []): txt = (getattr(line, "text", "") or "").strip() if not txt: continue conf = float(getattr(line, "confidence", 1.0)) bbox = getattr(line, "bbox", None) if bbox is None: bbox = getattr(line, "polygon", None) regions.append({ "text": txt, "confidence": round(conf, 4), "bbox": bbox, "word_count": len(txt.split()), }) weighted_total = sum(r["confidence"] * r["word_count"] for r in regions) weighted_words = sum(r["word_count"] for r in regions) conf = round(weighted_total / max(weighted_words, 1), 4) if regions else 0.0 return regions, conf def _run_tesseract_batch( images, langs: Optional[list[str]] = None, tessdata_dirs: Optional[list[Optional[str]]] = None, ): """ Tesseract fallback for OCR when Surya is unavailable/slow. Returns list of dicts with regions/confidence/method aligned to input order. """ try: import pytesseract except Exception as e: return [ { "regions": [], "confidence": 0.0, "method": f"error-no-tesseract ({e})", "tesseract_lang": (langs[idx] if langs and idx < len(langs) else "eng"), } for idx, _ in enumerate(images) ] # Prefer parallel image-level OCR with single-threaded internal OpenMP for better CPU utilization. os.environ.setdefault("OMP_THREAD_LIMIT", "1") def _ocr_single(img, lang_hint: str, tessdata_dir: Optional[str]): try: config = f"--oem 1 --psm {SS_TESSERACT_PSM}" if tessdata_dir: config = f"{config} --tessdata-dir \"{tessdata_dir}\"" data = pytesseract.image_to_data( img, lang=lang_hint or "eng", config=config, output_type=pytesseract.Output.DICT, timeout=SS_TESSERACT_TIMEOUT_SEC, ) regions = [] conf_weighted = 0.0 word_count = 0 n = len(data.get("text", [])) for i in range(n): txt = str(data["text"][i]).strip() if not txt: continue try: conf_raw = float(data["conf"][i]) except Exception: conf_raw = -1.0 if conf_raw < 0: continue conf = max(0.0, min(1.0, conf_raw / 100.0)) left = int(data["left"][i]) top = int(data["top"][i]) width = int(data["width"][i]) height = int(data["height"][i]) bbox = [left, top, left + width, top + height] wc = max(len(txt.split()), 1) regions.append({ "text": txt, "confidence": round(conf, 4), "bbox": bbox, "word_count": wc, }) conf_weighted += conf * wc word_count += wc avg_conf = round(conf_weighted / max(word_count, 1), 4) if regions else 0.0 # Hard gate for fallback quality: do not keep OCR text below minimum page confidence. if regions and avg_conf < SS_TESSERACT_MIN_CONF_KEEP: return { "regions": [], "confidence": 0.0, "method": "tesseract-lowconf-filtered", "tesseract_lang": lang_hint or "eng", } # Filter obvious OCR noise from illustration texture (common in no-text pages). full_text = " ".join(r["text"] for r in regions).strip() letters = sum(1 for c in full_text if c.isalpha()) printable = sum(1 for c in full_text if c.isprintable() and not c.isspace()) alpha_ratio = (letters / printable) if printable else 0.0 tokens = [t for t in full_text.split() if t] avg_token_len = (sum(len(t) for t in tokens) / len(tokens)) if tokens else 0.0 # Filter illustration noise — two thresholds: # 1. Strict: low conf + low alpha + short tokens (original heuristic, loosened) if regions and avg_conf <= 0.55 and alpha_ratio < 0.72 and avg_token_len < 3.5: return { "regions": [], "confidence": 0.0, "method": "tesseract-noise-filtered", "tesseract_lang": lang_hint or "eng", } # 2. Pure gibberish: very low alpha ratio regardless of confidence if regions and alpha_ratio < 0.50: return { "regions": [], "confidence": 0.0, "method": "tesseract-noise-filtered", "tesseract_lang": lang_hint or "eng", } # 3. Filter individual regions that look like illustration noise. # Key discriminator: real picture-book words avg 3.5+ chars. # Illustration noise (td, wht, LN, WA) avg 2.5 chars. clean_regions = [] for r in regions: txt = r["text"] tokens = [t for t in txt.split() if t and any(c.isalpha() for c in t)] if not tokens: continue r_avg_tok = sum(len(t) for t in tokens) / len(tokens) letters = sum(1 for c in txt if c.isalpha()) printable = sum(1 for c in txt if c.isprintable() and not c.isspace()) r_alpha = (letters / printable) if printable else 0.0 # Keep if: long enough tokens OR high confidence OR single short word (punctuation line) if r_avg_tok >= 3.2 or r["confidence"] >= 0.82 or (len(tokens) == 1 and r["confidence"] >= 0.65): clean_regions.append(r) if not clean_regions: # Everything filtered — page is pure illustration noise return { "regions": [], "confidence": 0.0, "method": "tesseract-noise-filtered", "tesseract_lang": lang_hint or "eng", } if len(clean_regions) < len(regions): # Recalculate confidence without noise regions cw = sum(r["confidence"] * r["word_count"] for r in clean_regions) wc = sum(r["word_count"] for r in clean_regions) avg_conf = round(cw / max(wc, 1), 4) regions = clean_regions return { "regions": regions, "confidence": avg_conf, "method": "tesseract", "tesseract_lang": lang_hint or "eng", } except RuntimeError as e: return { "regions": [], "confidence": 0.0, "method": f"error-tesseract-timeout ({e})", "tesseract_lang": lang_hint or "eng", } except Exception as e: return { "regions": [], "confidence": 0.0, "method": f"error-tesseract ({e})", "tesseract_lang": lang_hint or "eng", } if not images: return [] max_workers = min(SS_TESSERACT_WORKERS, len(images), max(1, os.cpu_count() or 1)) lang_list = list(langs) if langs else [] if len(lang_list) < len(images): lang_list.extend(["eng"] * (len(images) - len(lang_list))) elif len(lang_list) > len(images): lang_list = lang_list[: len(images)] tess_dir_list = list(tessdata_dirs) if tessdata_dirs else [] if len(tess_dir_list) < len(images): tess_dir_list.extend([None] * (len(images) - len(tess_dir_list))) elif len(tess_dir_list) > len(images): tess_dir_list = tess_dir_list[: len(images)] if max_workers <= 1: return [ _ocr_single(img, lang_list[idx], tess_dir_list[idx]) for idx, img in enumerate(images) ] outputs = [None] * len(images) executor = concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) futures = { executor.submit(_ocr_single, img, lang_list[idx], tess_dir_list[idx]): idx for idx, img in enumerate(images) } pending = set(futures.keys()) batch_timeout = max( SS_TESSERACT_BATCH_HARD_TIMEOUT_SEC, int((len(images) / max(max_workers, 1)) * SS_TESSERACT_TIMEOUT_SEC * 2 + 15), ) deadline = time.monotonic() + batch_timeout try: while pending and time.monotonic() < deadline: just_done, pending = concurrent.futures.wait( pending, timeout=0.75, return_when=concurrent.futures.FIRST_COMPLETED, ) if not just_done: continue for future in just_done: idx = futures[future] try: outputs[idx] = future.result() except Exception as e: outputs[idx] = {"regions": [], "confidence": 0.0, "method": f"error-tesseract-future ({e})"} if pending: for future in pending: idx = futures[future] outputs[idx] = { "regions": [], "confidence": 0.0, "method": f"error-tesseract-batch-timeout ({batch_timeout}s)", "tesseract_lang": lang_list[idx] if idx < len(lang_list) else "eng", } future.cancel() finally: # Avoid blocking the OCR run forever if one worker hangs in external OCR process. executor.shutdown(wait=False, cancel_futures=True) for i, out in enumerate(outputs): if out is None: outputs[i] = { "regions": [], "confidence": 0.0, "method": "error-tesseract-missing-output", "tesseract_lang": lang_list[i] if i < len(lang_list) else "eng", } return outputs def run_full_pipeline( files, rights_class: str, notes: str, book_code_hint: str, publication_year: str, scope_include: str, scope_exclude: str, safe_title: str, ): """ Single-button full pipeline: 1. Ingest + profile PDF 2. Update rights + save page scope 3. Run OCR Yields log updates throughout so UI stays live. """ log = [] def _line(msg): from datetime import datetime return f"[{datetime.now().strftime('%H:%M:%S')}] {msg}" # Step 1: Ingest + profile log.append(_line("① Registering PDF...")) yield _ingest_status_html("idle"), "\n".join(log), 0 try: status_html, manifest_df, ingest_log = ingest_pdfs( files, rights_class, notes, book_code_hint, publication_year ) for line in str(ingest_log).splitlines(): if line.strip(): log.append(line) except Exception as e: log.append(_line(f"✗ Ingest failed: {e}")) yield _ingest_status_html("idle"), "\n".join(log), 0 return yield status_html, "\n".join(log), 0 # Step 2: Save page scope + rights (resolve book_id from manifest) df = load_manifest_df() if df.empty: log.append(_line("✗ No books in manifest after ingest")) yield _ingest_status_html("idle"), "\n".join(log), 0 return # Get the book we just registered/updated latest_book_id = str(df.iloc[-1]["book_id"]) # Update rights if not unknown if rights_class and rights_class != "unknown": try: _, _, rights_log = update_rights(latest_book_id, rights_class) for line in str(rights_log).splitlines(): if line.strip() and "not found" not in line.lower(): log.append(line) except Exception as e: log.append(_line(f"⚠ Rights update: {e}")) # Save page scope if scope_include or scope_exclude or safe_title: log.append(_line(f"② Saving page scope for {latest_book_id}...")) yield _ingest_status_html("done"), "\n".join(log), 0 try: _, _, scope_log = save_book_scope( latest_book_id, scope_include, scope_exclude, safe_title ) for line in str(scope_log).splitlines(): if line.strip(): log.append(line) except Exception as e: log.append(_line(f"⚠ Scope save: {e}")) yield _ingest_status_html("done"), "\n".join(log), 0 # Step 3: Run OCR log.append(_line("③ Starting OCR...")) yield _ingest_status_html("done"), "\n".join(log), 0 try: ocr_gen = run_ocr(scope_include, scope_exclude, replace_book_queue=True) for ocr_status_html, ocr_log, _ocr_done in ocr_gen: for line in str(ocr_log).splitlines(): if line.strip() and line not in log: log.append(line) yield _ingest_status_html("done"), "\n".join(log), 0 except Exception as e: import traceback log.append(_line(f"✗ OCR error: {e}")) log.append(traceback.format_exc()) yield _ingest_status_html("done"), "\n".join(log), 0 return log.append(_line("✓ Pipeline complete — go to Review tab")) yield _ingest_status_html("done"), "\n".join(log), 1 # 1 = triggers review load def run_ocr(page_selection: str = "", page_exclusion: str = "", replace_book_queue: bool = True) -> tuple: """Run Surya OCR on all profiled PDFs.""" df = load_manifest_df() debug = f"[DEBUG] SS_ROOT={SS_ROOT}\nMANIFEST_CSV={MANIFEST_CSV}\nCSV exists={MANIFEST_CSV.exists()}\n" if not df.empty: debug += f"Manifest rows={len(df)}\nStatuses={df['status'].value_counts().to_dict()}\n" try: row_summaries = [] for _, r in df.iterrows(): row_summaries.append( f"{r.get('book_id','?')} rights={r.get('rights_class','?')} status={r.get('status','?')}" ) if row_summaries: debug += "Rows:\n- " + "\n- ".join(row_summaries) + "\n" except Exception: pass else: debug += "Manifest is EMPTY\n" if df.empty: return _ocr_status_html(), debug + "No sources. Complete Steps 1-2 first." eligible = df[df["status"].isin(["profiled", "ocred", "rendered"])] if eligible.empty: unknown_or_excluded = df[df["rights_class"].isin(["unknown", "excluded"])] if "rights_class" in df.columns else df.iloc[0:0] pending = df[df["status"] == "pending"] if "status" in df.columns else df.iloc[0:0] debug += ( f"Eligible rows={len(eligible)}\n" f"Pending rows={len(pending)}\n" f"Unknown/excluded rights={len(unknown_or_excluded)}\n" "Tip: In Step 1, update rights_class away from unknown/excluded, " "then rerun Step 2 Profile.\n" ) return _ocr_status_html(), debug + "No profiled PDFs. Complete Step 2 first." log = [] queue_rows = [] processed_books = [] selection_set, selection_error = _parse_page_selection(page_selection) if selection_error: return _ocr_status_html(), f"{debug}Invalid page selection: {selection_error}" manual_selection = selection_set is not None exclusion_spec = (page_exclusion or "").strip().lower() exclusion_request = None if exclusion_spec in ("", "none", "-") else exclusion_spec manual_exclusion = exclusion_request is not None if exclusion_request is not None: exclusion_set, exclusion_error = _parse_page_selection(exclusion_request) if exclusion_error: return _ocr_status_html(), f"{debug}Invalid exclusion selection: {exclusion_error}" else: exclusion_set = set() if selection_set is None: log.append(log_line("ℹ Page selection: all pages")) else: selected_preview = ",".join(str(p) for p in sorted(selection_set)) log.append(log_line(f"ℹ Page selection: {selected_preview}")) if exclusion_request is None: log.append(log_line("ℹ Page exclusion: none")) elif exclusion_set is None: log.append(log_line("ℹ Page exclusion: all selected pages")) else: excluded_preview = ",".join(str(p) for p in sorted(exclusion_set)) log.append(log_line(f"ℹ Page exclusion: {excluded_preview}")) surya, surya_error, reused = _get_surya_runtime_with_timeout(SS_SURYA_LOAD_TIMEOUT_SEC) if surya: if reused: log.append(log_line(f"✓ Reusing Surya models ({surya['api']})")) else: log.append(log_line(f"✓ Surya models loaded ({surya['api']})")) else: log.append(log_line(f"⚠ Surya unavailable ({surya_error}) — using Tesseract fallback for this run")) log.append( log_line( f"ℹ Tesseract fallback config: workers={SS_TESSERACT_WORKERS} timeout={SS_TESSERACT_TIMEOUT_SEC}s " f"psm={SS_TESSERACT_PSM} dpi={SS_RENDER_DPI_FALLBACK} min_conf_keep={SS_TESSERACT_MIN_CONF_KEEP:.2f}" ) ) yield _ocr_status_html(), "\n".join(log), 0 # stream: models loaded punct_mod = _load_punct_module() if punct_mod is None: log.append(log_line(f"⚠ Punctuation corrector unavailable ({_PUNCT_MODULE_ERROR or 'unknown'})")) else: log.append(log_line("✓ Punctuation corrector active (rule-check + confidence penalty + learning map)")) font_mod = _load_font_module() mixfont_enabled = False if font_mod is None: log.append(log_line(f"⚠ Font library unavailable ({_FONT_MODULE_ERROR or 'unknown'})")) else: preflight = _mixfont_preflight() if preflight.get("api_key_set") and preflight.get("public_image_url_source_available"): log.append(log_line("✓ MixFont detection enabled (per-page font routing)")) mixfont_enabled = True else: missing = [] if not preflight.get("api_key_set"): missing.append("MIXFONT_API_KEY") if not preflight.get("public_image_url_source_available"): missing.append("MIXFONT_IMAGE_BASE_URL (or MIXFONT_IMAGE_URL_TEMPLATE/SPACE_HOST)") missing_text = ", ".join(missing) if missing else "unknown config" log.append(log_line(f"ℹ MixFont disabled this run — missing {missing_text}")) for _, row in eligible.iterrows(): book_id = row["book_id"] log.append(log_line(f"▶ {book_id}: starting OCR")) yield _ocr_status_html(), "\n".join(log), 0 # stream: book start profile_path = PROFILES_DIR / f"{book_id}_page_profile.json" if not profile_path.exists(): log.append(log_line(f"✗ {book_id}: no profile")) continue with open(profile_path, encoding="utf-8") as f: profile = json.load(f) pages_all = profile.get("pages", []) page_numbers = [int(p.get("page_number", 0) or 0) for p in pages_all] max_page = max(page_numbers) if page_numbers else None if selection_set is None: selected_pages_for_book = set(page_numbers) else: selected_pages_for_book, sel_err = _parse_page_selection(page_selection, max_page=max_page) if sel_err: log.append(log_line(f"⚠ {book_id}: {sel_err}")) continue if selected_pages_for_book is None: selected_pages_for_book = set(page_numbers) if exclusion_request is None: excluded_pages_for_book: set[int] = set() else: excluded_pages_for_book, excl_err = _parse_page_selection(exclusion_request, max_page=max_page) if excl_err: log.append(log_line(f"⚠ {book_id}: {excl_err}")) continue if excluded_pages_for_book is None: excluded_pages_for_book = set(page_numbers) saved_include_spec = _normalize_saved_spec(row.get("story_pages_include", "")) if saved_include_spec and not manual_selection: saved_include_set, saved_inc_err = _parse_page_selection(saved_include_spec, max_page=max_page) if saved_inc_err: log.append(log_line(f"⚠ {book_id}: invalid saved include pages '{saved_include_spec}' ({saved_inc_err})")) elif saved_include_set is not None: selected_pages_for_book = set(selected_pages_for_book) & set(saved_include_set) saved_exclude_spec = _normalize_saved_spec(row.get("story_pages_exclude", "")) if saved_exclude_spec and not manual_exclusion: saved_exclude_set, saved_exc_err = _parse_page_selection(saved_exclude_spec, max_page=max_page) if saved_exc_err: log.append(log_line(f"⚠ {book_id}: invalid saved exclude pages '{saved_exclude_spec}' ({saved_exc_err})")) elif saved_exclude_set is not None: excluded_pages_for_book = set(excluded_pages_for_book) | set(saved_exclude_set) selected_pages_for_book = set(selected_pages_for_book) - set(excluded_pages_for_book) if not selected_pages_for_book: log.append(log_line(f"⚠ {book_id}: no pages left after applying selection/exclusion")) continue if saved_include_spec and not manual_selection: log.append(log_line(f"ℹ {book_id}: saved include pages {saved_include_spec}")) if saved_exclude_spec and not manual_exclusion: log.append(log_line(f"ℹ {book_id}: saved exclude pages {saved_exclude_spec}")) selected_preview = ",".join(str(p) for p in sorted(selected_pages_for_book)) log.append(log_line(f"ℹ {book_id}: effective selected pages {selected_preview}")) processed_books.append(book_id) ocr_pages = [] review_pages = [] quarantine_pages = [] cal = load_calibration() default_cal = cal.get("_default", DEFAULT_CALIBRATION["_default"]) doc = None try: import fitz pdf_path = SOURCE_DIR / row["filename"] doc = fitz.open(str(pdf_path)) except Exception as e: log.append(log_line(f"⚠ {book_id}: PDF open failed ({e})")) # First pass: render OCR/hybrid pages once and keep PIL images for batch OCR. ocr_targets = [] render_dpi = SS_RENDER_DPI_SURYA if surya is not None else SS_RENDER_DPI_FALLBACK book_detected_font = None mixfont_attempts = 0 mixfont_errors = 0 mixfont_disabled_for_book = not mixfont_enabled for page_data in profile.get("pages", []): page_num = page_data["page_number"] route = page_data["route"] font_name = page_data.get("font_name") if selected_pages_for_book is not None and int(page_num) not in selected_pages_for_book: continue if route == "embedded_text": continue render_path = None if doc is not None: try: existing_rel = page_data.get("render_path") existing_dpi = int(page_data.get("render_dpi", 0) or 0) existing_abs = (SS_ROOT / existing_rel) if existing_rel else None if existing_abs and existing_abs.exists() and existing_dpi == render_dpi: render_path = existing_abs else: page = doc[page_num - 1] render_start = time.time() # Always render greyscale — better OCR, consistent review display pix = page.get_pixmap( dpi=render_dpi, alpha=False, annots=False, colorspace=fitz.csGRAY, ) render_dir = RENDERS_DIR / book_id render_dir.mkdir(exist_ok=True) render_path = render_dir / f"{book_id}_page_{page_num:04d}_{render_dpi}dpi.png" pix.save(str(render_path)) render_secs = round(time.time() - render_start, 2) if render_secs >= 4.0: log.append(log_line(f" ⚠ {book_id} p{page_num}: slow render {render_secs}s at {render_dpi}dpi")) page_data["render_path"] = str(render_path.relative_to(SS_ROOT)) page_data["render_dpi"] = render_dpi except Exception as e: log.append(log_line(f" ⚠ {book_id} p{page_num}: render failed ({e})")) font_name = page_data.get("font_name") if (not font_name) and book_detected_font: page_data["font_name"] = book_detected_font font_name = book_detected_font if ( (not font_name) and render_path is not None and (not mixfont_disabled_for_book) and mixfont_attempts < SS_MIXFONT_MAX_DETECT_PAGES ): mixfont_attempts += 1 font_result = _identify_page_font(str(render_path)) detected_font = font_result.get("font_name") if detected_font: page_data["font_name"] = detected_font page_data["font_confidence"] = float(font_result.get("confidence", 0.0) or 0.0) page_data["font_detected_at"] = datetime.utcnow().isoformat() + "Z" book_detected_font = detected_font font_name = detected_font log.append(log_line(f"ℹ {book_id}: detected font '{detected_font}'")) else: page_data["font_name"] = None page_data["font_detect_error"] = font_result.get("error") mixfont_errors += 1 if mixfont_errors >= SS_MIXFONT_MAX_ERRORS: mixfont_disabled_for_book = True err_txt = page_data.get("font_detect_error") or "unknown" log.append(log_line(f"⚠ {book_id}: MixFont disabled for this run ({err_txt})")) ocr_targets.append({ "page_num": page_num, "route": route, "render_path": page_data.get("render_path"), "font_name": page_data.get("font_name"), }) # Batch OCR for non-embedded pages. ocr_lookup = {} if ocr_targets: batch_size = SS_SURYA_BATCH_SIZE for start in range(0, len(ocr_targets), batch_size): batch = ocr_targets[start:start + batch_size] batch_pages = [item["page_num"] for item in batch] batch_images = [] batch_items_with_images = [] batch_tess_langs = [] batch_tess_dirs = [] try: from PIL import Image except Exception as e: Image = None log.append(log_line(f" ⚠ PIL unavailable for OCR batch ({e})")) if Image is not None: for item in batch: rel_path = item.get("render_path") if not rel_path: ocr_lookup[item["page_num"]] = { "regions": [], "confidence": 0.0, "method": "error-no-render-path", } continue render_abs = SS_ROOT / rel_path if not render_abs.exists(): ocr_lookup[item["page_num"]] = { "regions": [], "confidence": 0.0, "method": "error-render-missing", } continue try: img = Image.open(render_abs).convert("RGB") batch_images.append(img) batch_items_with_images.append(item) lang_model = _resolve_tesseract_lang(item.get("font_name")) batch_tess_langs.append(lang_model) batch_tess_dirs.append(_resolve_tessdata_dir(lang_model)) except Exception as e: ocr_lookup[item["page_num"]] = { "regions": [], "confidence": 0.0, "method": f"error-open-image ({e})", } try: if not batch_items_with_images: raise RuntimeError("No render images available in this OCR batch.") if surya is not None: predictions = _run_surya_batch(batch_images, surya) for item, page_result in zip(batch_items_with_images, predictions): regions, conf = _regions_from_page_result(page_result) ocr_lookup[item["page_num"]] = { "regions": regions, "confidence": conf, "method": "surya", } else: fallback_preds = _run_tesseract_batch(batch_images, batch_tess_langs, batch_tess_dirs) timeout_count = sum(1 for pred in fallback_preds if "batch-timeout" in str(pred.get("method", ""))) for item, pred in zip(batch_items_with_images, fallback_preds): ocr_lookup[item["page_num"]] = pred if timeout_count: log.append( log_line( f" ⚠ {book_id} batch {batch_pages[0]}-{batch_pages[-1]}: " f"{timeout_count}/{len(fallback_preds)} tesseract timeouts" ) ) except Exception as e: # If Surya batch fails, try Tesseract for this batch before giving up. if surya is not None: log.append(log_line(f" ⚠ {book_id} batch {batch_pages[0]}-{batch_pages[-1]} Surya error: {e}; retrying with Tesseract")) fallback_preds = _run_tesseract_batch(batch_images, batch_tess_langs, batch_tess_dirs) timeout_count = sum(1 for pred in fallback_preds if "batch-timeout" in str(pred.get("method", ""))) for item, pred in zip(batch_items_with_images, fallback_preds): ocr_lookup[item["page_num"]] = pred if timeout_count: log.append( log_line( f" ⚠ {book_id} batch {batch_pages[0]}-{batch_pages[-1]} retry: " f"{timeout_count}/{len(fallback_preds)} tesseract timeouts" ) ) else: for item in batch_items_with_images: ocr_lookup[item["page_num"]] = { "regions": [], "confidence": 0.0, "method": "error", } log.append(log_line(f" ⚠ {book_id} batch {batch_pages[0]}-{batch_pages[-1]}: {e}")) finally: for img in batch_images: try: img.close() except Exception: pass # Stream progress after each batch done_pages = min(start + batch_size, len(ocr_targets)) log.append(log_line(f" ✓ {book_id}: batch {start//batch_size + 1} complete ({done_pages}/{len(ocr_targets)} pages)")) yield _ocr_status_html(), "\n".join(log), 0 log.append(log_line(f" → {book_id}: batch OCR complete — building review queue")) yield _ocr_status_html(), "\n".join(log), 0 # stream: batch OCR done # Second pass: build page-level OCR output and review queue. for page_data in profile.get("pages", []): page_num = page_data["page_number"] route = page_data["route"] font_name = page_data.get("font_name") # reset per page — prevents stale carry-over if selected_pages_for_book is not None and int(page_num) not in selected_pages_for_book: continue if route == "embedded_text": try: if doc is None: raise RuntimeError("PDF document not open") text = doc[page_num - 1].get_text("text").strip() regions = [{"text": text, "confidence": 0.99, "bbox": [0, 0, 100, 100], "word_count": len(text.split())}] conf = 0.99 method = "embedded-text" except Exception: regions = [] conf = 0.0 method = "error" elif surya or page_num in ocr_lookup: page_out = ocr_lookup.get(page_num, {"regions": [], "confidence": 0.0, "method": "error"}) regions = page_out["regions"] conf = page_out["confidence"] method = page_out["method"] else: regions = [] conf = 0.0 method = "skipped-no-surya" raw_text = " ".join(r["text"] for r in regions)[:500] # Skip pages matching known noise patterns for this book _noise_pats = _load_noise_patterns(book_id) if _noise_pats and _text_matches_noise(raw_text, _noise_pats): log.append(log_line(f" ⊘ {book_id} p{page_num}: matched noise pattern — skipped")) continue corrected_text, punct_flags, punct_score = _apply_punctuation_corrections( raw_text, book_id, font_name=font_name, ) conf_adjusted = _punctuation_confidence_penalty(corrected_text, conf) if font_name: if str(method).startswith("tesseract"): _update_font_engine_stats(font_name, 0.0, conf) elif str(method).startswith("surya"): _update_font_engine_stats(font_name, conf, 0.0) if method in ("tesseract-noise-filtered", "tesseract-lowconf-filtered") and not regions: conf_class = "auto-accept" elif conf_adjusted >= default_cal["auto_accept"]: conf_class = "auto-accept" elif conf_adjusted >= default_cal["review"]: conf_class = "review-required" elif conf_adjusted >= default_cal["quarantine"]: conf_class = "low-confidence" else: conf_class = "quarantine" # If punctuation rules detect likely text-quality issues, force at least review-required. if punct_flags and conf_class == "auto-accept": conf_class = "review-required" if conf_class in ("review-required", "low-confidence"): review_pages.append(page_num) elif conf_class == "quarantine": quarantine_pages.append(page_num) ocr_pages.append({ "page_number": page_num, "route": route, "regions": regions, "page_confidence_raw": conf, "page_confidence": conf_adjusted, "confidence_class": conf_class, "extraction_method": method, "punctuation_score": punct_score, "punctuation_flags": punct_flags, "font_name": font_name, "tesseract_lang": ocr_lookup.get(page_num, {}).get("tesseract_lang"), "render_path": page_data.get("render_path"), "ocred_at": datetime.utcnow().isoformat() + "Z", }) if conf_class in ("review-required", "low-confidence", "quarantine"): region_id = f"{book_id}_p{page_num:04d}" queue_rows.append({ "book_id": book_id, "filename": row["filename"], "page": page_num, "region_id": region_id, "region_class": "narration", "crop_path": page_data.get("render_path", ""), "raw_ocr": corrected_text, "raw_ocr_original": raw_text, "confidence_raw": conf, "confidence": conf_adjusted, "confidence_class": conf_class, "punct_score": punct_score, "punct_flags_count": len(punct_flags), "font_name": font_name or "", "tesseract_lang": ocr_lookup.get(page_num, {}).get("tesseract_lang", ""), "status": "quarantine" if conf_class == "quarantine" else "pending", "reviewer": "", "correction": "", "reason_code": "", }) if doc is not None: try: doc.close() except Exception: pass ocr_dir = OCR_RAW_DIR / book_id ocr_dir.mkdir(exist_ok=True) with open(ocr_dir / f"{book_id}_ocr_raw.json", "w", encoding="utf-8") as f: json.dump({ "book_id": book_id, "filename": row["filename"], "source_hash": row["sha256"], "page_count": len(ocr_pages), "review_queue": review_pages, "quarantine_list": quarantine_pages, "pages": ocr_pages, }, f, indent=2) with open(profile_path, "w", encoding="utf-8") as f: json.dump(profile, f, indent=2) df.loc[df["book_id"] == book_id, "status"] = "ocred" total_review = len(review_pages) + len(quarantine_pages) log.append(log_line(f"✓ {book_id}: {len(ocr_pages)} selected pp — review queue: {total_review} · queue write complete")) yield _ocr_status_html(), "\n".join(log), 0 # stream: book done qdf_new = pd.DataFrame(queue_rows) if QUEUE_CSV.exists(): existing = pd.read_csv(QUEUE_CSV) else: existing = pd.DataFrame() if replace_book_queue and processed_books and not existing.empty and "book_id" in existing.columns: existing = existing[~existing["book_id"].isin(processed_books)] if not existing.empty and not qdf_new.empty: qdf_out = pd.concat([existing, qdf_new], ignore_index=True).drop_duplicates(subset=["region_id"], keep="last") elif not qdf_new.empty: qdf_out = qdf_new else: qdf_out = existing if not qdf_out.empty: qdf_out.to_csv(QUEUE_CSV, index=False) elif QUEUE_CSV.exists(): QUEUE_CSV.unlink() save_manifest_df(df) yield _ocr_status_html(), "\n".join(log), 1 # stream: final — triggers review load def _ocr_status_html() -> str: q_df = load_queue_df() d_df = load_decisions_df() total_queue = len(q_df) decided = len(d_df) pending = total_queue - decided auto = len(q_df[q_df["confidence_class"] == "auto-accept"]) if not q_df.empty and "confidence_class" in q_df.columns else 0 quar = len(q_df[q_df["confidence_class"] == "quarantine"]) if not q_df.empty and "confidence_class" in q_df.columns else 0 cal = load_calibration() default = cal.get("_default", DEFAULT_CALIBRATION["_default"]) corrections = sum(v.get("corrections",0) for v in cal.values() if isinstance(v, dict)) punct_pairs = int(_punctuation_map_summary().get("total_pairs", 0)) training_badge = ( f'
' f'{corrections} corrections logged · punct map={punct_pairs} pairs · ' f'thresholds auto={default["auto_accept"]:.0%} review={default["review"]:.0%}
' ) return f"""
{total_queue}
Review Queue
{pending}
Pending
{quar}
Quarantined
{decided}
Decided
{training_badge}
""" # ── Step 4: REVIEW ───────────────────────────────────────────────────────────── def get_review_item(idx: int) -> tuple: q_df = load_queue_df() d_df = load_decisions_df() if q_df.empty: return None, "", "", 0, 0, "" decided_ids = set(d_df["region_id"].tolist()) if not d_df.empty else set() pending = q_df[~q_df["region_id"].isin(decided_ids)] if not pending.empty and {"book_id", "page"}.issubset(pending.columns): pending = pending.sort_values(["book_id", "page"], ascending=[True, True], kind="stable") if pending.empty: return None, "All items reviewed!", "", len(q_df), len(q_df), "" idx = idx % len(pending) item = pending.iloc[idx] img_path = None crop_path = item.get("crop_path","") if crop_path: candidate = SS_ROOT / crop_path if candidate.exists(): img_path = str(candidate) else: # /tmp wiped after restart — re-render page from PDF on demand try: import fitz book_id = item.get("book_id","") page_num = int(item.get("page", 1)) pdf_path = SOURCE_DIR / f"{book_id}.pdf" if not pdf_path.exists(): # try original filename from manifest mdf = pd.read_csv(MANIFEST_CSV) row = mdf[mdf["book_id"] == book_id] if not row.empty: pdf_path = SOURCE_DIR / row.iloc[0]["filename"] if pdf_path.exists(): candidate.parent.mkdir(parents=True, exist_ok=True) doc = fitz.open(str(pdf_path)) pg = doc[page_num - 1] pix = pg.get_pixmap(dpi=150) pix.save(str(candidate)) doc.close() img_path = str(candidate) except Exception: img_path = None font_suffix = f" · font: {item.get('font_name')}" if str(item.get("font_name", "")).strip() else "" info = (f"
" f"{item['book_id']} · page {item['page']}{font_suffix} · " f"conf: " f"{float(item.get('confidence',0)):.0%}
") raw_font = item.get("font_name", "") or "" font_name_val = "" if str(raw_font).lower() in ("nan", "none", "") else str(raw_font).strip() return img_path, item.get("raw_ocr",""), info, len(q_df) - len(pending), len(q_df), font_name_val def save_review_decision(idx: int, final_text: str, action: str, noise_text_input: str, reason: str, conf_override: bool = False) -> tuple: q_df = load_queue_df() d_df = load_decisions_df() if q_df.empty: return "No queue.", *get_review_item(idx)[1:] decided_ids = set(d_df["region_id"].tolist()) if not d_df.empty else set() pending = q_df[~q_df["region_id"].isin(decided_ids)] if not pending.empty and {"book_id", "page"}.issubset(pending.columns): pending = pending.sort_values(["book_id", "page"], ascending=[True, True], kind="stable") if pending.empty: return "All done!", *get_review_item(0)[1:] idx = idx % len(pending) item = pending.iloc[idx] raw_text = item.get("raw_ocr", "") raw_text_original = item.get("raw_ocr_original", raw_text) was_correct = (final_text.strip() == raw_text.strip()) region_class = item.get("region_class","narration") # Recalibrate confidence thresholds recalibrate(region_class, was_correct, float(item.get("confidence",0))) # Save decision decision = { "region_id": item["region_id"], "book_id": item["book_id"], "page": item["page"], "status": action, "final_text": final_text, "raw_text": raw_text, "raw_text_original": raw_text_original, "reason_code": reason, "reviewer": "smoke-signal", "font_name": item.get("font_name", ""), "was_correct": was_correct, "decided_at": datetime.utcnow().isoformat() + "Z", } # Append to decisions CSV new_row = pd.DataFrame([decision]) if DECISIONS_CSV.exists(): d_df = pd.concat([d_df, new_row], ignore_index=True) else: d_df = new_row d_df.to_csv(DECISIONS_CSV, index=False) # Append to gold training set try: punct_score = float(item.get("punct_score", 1.0)) except Exception: punct_score = 1.0 try: punct_flags_count = int(float(item.get("punct_flags_count", 0))) except Exception: punct_flags_count = 0 final_confidence = 1.0 if conf_override else float(item.get("confidence", 0)) if conf_override: decision["confidence_override"] = True with open(GOLD_FILE, "a", encoding="utf-8") as f: f.write(json.dumps({ **decision, "region_class": region_class, "confidence": final_confidence, "conf_class": "verified-100" if conf_override else item.get("confidence_class",""), "punct_score": punct_score, "punct_flags_count": punct_flags_count, }, ensure_ascii=False, default=_json_default) + "\n") learned_pairs = 0 learned_noise_added = 0 active_noise_patterns = 0 if action == "rejected" and str(reason).strip() == "NON_STORY_TEXT": # Use pasted noise text if provided, otherwise fall back to raw OCR noise_source = str(noise_text_input or "").strip() or final_text.strip() or raw_text.strip() book_id_for_noise = str(item.get("book_id", "")) for line in noise_source.splitlines(): line = line.strip() if line: if _save_noise_pattern(book_id_for_noise, line): learned_noise_added += 1 if learned_noise_added == 0 and noise_source: if _save_noise_pattern(book_id_for_noise, noise_source): learned_noise_added += 1 active_noise_patterns = len(_load_noise_patterns(book_id_for_noise)) if action == "edited": learned_pairs = _record_punctuation_correction( raw_text_original, final_text, item.get("book_id", ""), font_name=item.get("font_name"), ) cal = load_calibration() default = cal.get("_default", DEFAULT_CALIBRATION["_default"]) corrections = sum(v.get("corrections",0) for v in cal.values() if isinstance(v,dict)) gold_count = sum(1 for _ in open(GOLD_FILE)) if GOLD_FILE.exists() else 0 punct_pairs = int(_punctuation_map_summary().get("total_pairs", 0)) learned_note = f" · +{learned_pairs} punct learns" if learned_pairs else "" noise_note = "" if action == "rejected" and str(reason).strip() == "NON_STORY_TEXT": noise_note = f" · +{learned_noise_added} noise learns · active noise patterns: {active_noise_patterns}" feedback = (f"
" f"Gold set: {gold_count} examples · {corrections} corrections · punct map: {punct_pairs} pairs" f"{learned_note}{noise_note} · auto-accept threshold: {default['auto_accept']:.0%}
") return feedback, *get_review_item(0) def _review_status_html() -> str: q_df = load_queue_df() d_df = load_decisions_df() total = len(q_df) decided = len(d_df) pending = total - decided accepted = len(d_df[d_df["status"] == "accepted"]) if not d_df.empty else 0 edited = len(d_df[d_df["status"] == "edited"]) if not d_df.empty else 0 rejected = len(d_df[d_df["status"] == "rejected"]) if not d_df.empty else 0 gold_count = sum(1 for _ in open(GOLD_FILE)) if GOLD_FILE.exists() else 0 pct = int(decided / max(total, 1) * 100) return f"""
{pending}
Pending
{accepted + edited}
Approved
{rejected}
Rejected
{gold_count}
Gold Examples
""" # ── Step 5: EXPORT ───────────────────────────────────────────────────────────── def run_export() -> tuple: """Export approved decisions to Codex JSONL + gold set.""" d_df = load_decisions_df() q_df = load_queue_df() df = load_manifest_df() if d_df.empty: return _export_status_html(), "No review decisions. Complete Step 4 first.", None, None approved = d_df[d_df["status"].isin(["accepted","edited"])] if approved.empty: return _export_status_html(), "No approved items to export.", None, None log = [] records = [] batch = ts() for _, dec in approved.iterrows(): book_id = dec["book_id"] manifest_row = df[df["book_id"] == book_id].iloc[0] if not df[df["book_id"] == book_id].empty else {} records.append({ "book_id": book_id, "safe_title": manifest_row.get("safe_title","") if isinstance(manifest_row, pd.Series) else "", "source_hash": manifest_row.get("sha256","") if isinstance(manifest_row, pd.Series) else "", "page_number": dec["page"], "region_id": dec["region_id"], "text_final": dec["final_text"], "text_raw": dec.get("raw_text",""), "review_status": dec["status"], "reviewer": dec.get("reviewer",""), "extraction_method": "ocr", "export_batch": batch, "exported_at": datetime.utcnow().isoformat() + "Z", }) # Write Codex JSONL jsonl_path = EXPORTS_DIR / f"codex_export_{batch}.jsonl" with open(jsonl_path, "w", encoding="utf-8") as f: for r in records: f.write(json.dumps(r, ensure_ascii=False) + "\n") # Write gold set copy gold_export = EXPORTS_DIR / f"gold_set_{batch}.jsonl" if GOLD_FILE.exists(): import shutil shutil.copy2(GOLD_FILE, gold_export) log.append(log_line(f"✓ Exported {len(records)} approved records → {jsonl_path.name}")) log.append(log_line(f"✓ Gold training set → {gold_export.name}")) log.append(log_line(f"✓ Ready for Codex ingestion")) return _export_status_html(len(records)), "\n".join(log), str(jsonl_path), str(gold_export) def _export_status_html(last_export=0) -> str: exports = list(EXPORTS_DIR.glob("codex_export_*.jsonl")) total_exports = len(exports) gold_count = sum(1 for _ in open(GOLD_FILE)) if GOLD_FILE.exists() else 0 d_df = load_decisions_df() approved = len(d_df[d_df["status"].isin(["accepted","edited"])]) if not d_df.empty else 0 return f"""
{approved}
Approved
{last_export or '—'}
Last Export
{total_exports}
Export Batches
{gold_count}
Gold Examples
""" def _wizard_header_html(active_step: int = 1) -> str: step_labels = [ "INGEST+PROFILE", "PROFILE (OPTIONAL)", "OCR", "REVIEW", "EXPORT", ] active_step = max(1, min(5, int(active_step or 1))) step_html = [] for idx, label in enumerate(step_labels, start=1): cls = "ss-step active" if idx == active_step else "ss-step" step_html.append(f'
{idx}{label}
') if idx < len(step_labels): step_html.append('
') return """
""" + "".join(step_html) + """
""" # ── Main tab builder ─────────────────────────────────────────────────────────── def smoke_signal_tab(): """Call this inside your gr.Blocks() Tabs to add the Smoke Signal tab.""" with gr.TabItem("◈ Smoke Signal", elem_id="ss-tab"): # ── Wizard header ──────────────────────────────────────────────────── wizard_header = gr.HTML(_wizard_header_html(1)) with gr.Tabs() as wizard: # ── STEP 1: INGEST ──────────────────────────────────────────────── with gr.TabItem("① Ingest", id="ss-ingest") as ingest_tab: gr.HTML("""
📥

Text Extraction

Drop PDF · Define story pages · Extract clean text for author fingerprinting

""") ingest_status = gr.HTML(_ingest_status_html("idle")) with gr.Row(): with gr.Column(scale=2): pdf_upload = gr.File( label="① Drop PDF here", file_types=[".pdf"], file_count="multiple", type="filepath", ) with gr.Column(scale=1): scope_include = gr.Textbox( label="② Story Pages (include)", placeholder="e.g. 7-27 or 7,8,9,11-27 or all", lines=1, ) scope_exclude = gr.Textbox( label="Story Pages (exclude)", placeholder="e.g. 1,2,3,17,25 — title/copyright pages", lines=1, ) # Hidden fields — auto-populated, not shown to user rights_dd = gr.Dropdown(choices=["licensed-owned"], value="licensed-owned", visible=False) book_code_input = gr.Textbox(visible=False, value="") pub_year_input = gr.Textbox(visible=False, value="") ingest_notes = gr.Textbox(visible=False, value="") scope_title = gr.Textbox(visible=False, value="") run_pipeline_btn = gr.Button( "③ Extract Text →", elem_classes=["ss-btn-run"], ) pipeline_done = gr.State(0) ingest_log = gr.Textbox(label="Log", lines=10, interactive=False, elem_classes=["ss-log"]) ingest_copy_log_btn = gr.Button("Copy Log", size="sm") manifest_table = gr.DataFrame( label="Source Manifest", interactive=False, wrap=True, ) refresh_btn = gr.Button("↻ Refresh Manifest", size="sm") # Keep hidden compat fields for wiring that references them later update_book_id = gr.Textbox(visible=False, value="") scope_book_id = gr.Textbox(visible=False, value="") update_rights_dd = gr.Dropdown(choices=["public-domain","licensed-owned","controlled-internal","unknown"], value="licensed-owned", visible=False) scope_save_btn = gr.Button(visible=False) ingest_btn = gr.Button(visible=False) # ── Auto-populate: book code + year → both book ID fields ────── def _make_book_id(code, year): code = (code or "").strip().lower()[:3] year = (year or "").strip()[:4] if code and year: return f"{code}{year}", f"{code}{year}" return "", "" book_code_input.change( _make_book_id, inputs=[book_code_input, pub_year_input], outputs=[update_book_id, scope_book_id], ) pub_year_input.change( _make_book_id, inputs=[book_code_input, pub_year_input], outputs=[update_book_id, scope_book_id], ) # ── Auto-populate: PDF filename → safe book title ───────────── def _auto_fill_from_upload(files): """Auto-generate book code, year, and safe title from filename.""" import re as _re if not files: return "", "", "" first = files[0] if isinstance(files, list) else files name = Path(first).stem if isinstance(first, str) else "" slug = _re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") # Extract year if present in filename (e.g. "gruffalo-1999") year_match = _re.search(r"(19|20)\d{2}", name) year = year_match.group(0) if year_match else "" # Book code: first 3 alpha chars of filename alpha = _re.sub(r"[^a-z]", "", name.lower()) code = alpha[:3] if len(alpha) >= 3 else alpha.ljust(3, "x") return code, year, slug pdf_upload.change( _auto_fill_from_upload, inputs=[pdf_upload], outputs=[book_code_input, pub_year_input, scope_title], ) # Main pipeline button — does everything run_pipeline_btn.click( run_full_pipeline, inputs=[pdf_upload, rights_dd, ingest_notes, book_code_input, pub_year_input, scope_include, scope_exclude, scope_title], outputs=[ingest_status, ingest_log, pipeline_done], show_progress="full", trigger_mode="once", ) # Refresh manifest after pipeline completes pipeline_done.change(lambda: load_manifest_df(), outputs=[manifest_table]) refresh_btn.click(lambda: load_manifest_df(), outputs=[manifest_table]) ingest_copy_log_btn.click( fn=None, inputs=[ingest_log], outputs=[], js="(logTxt) => { if (navigator && navigator.clipboard) { navigator.clipboard.writeText(logTxt || ''); } return []; }", ) gr.HTML('
') # Auto-load removed — use Refresh button instead to avoid SSR hang # ── STEP 2: PROFILE ─────────────────────────────────────────────── # ── STEP 2: OCR ─────────────────────────────────────────────────── with gr.TabItem("② OCR", id="ss-ocr") as ocr_tab: gr.HTML("""
👁

OCR Engine

Surya layout + recognition · Confidence scoring · Review queue

▶ Order: Ingest + Save Book Pages   Click Run OCR below   Wait for log to complete   Go to Review tab
""") ocr_status = gr.HTML(_ocr_status_html()) ocr_done = gr.State(0) # increments when OCR completes — triggers review load gr.HTML("""
Self-Improvement Loop
Every correction you make in Step 4 is logged to the gold training set and recalibrates the confidence thresholds for that region class in real time. The more you review, the smarter the pipeline gets — without retraining.
""") ocr_page_selection = gr.Textbox( label="Pages to OCR (optional)", placeholder="all or e.g. 7 or 3-8 or 1,4,9-12", lines=1, ) ocr_page_exclusion = gr.Textbox( label="Pages to Exclude (optional)", placeholder="e.g. 1,2,3,17,25 (PDF page numbers)", lines=1, ) ocr_replace_queue = gr.Checkbox( label="Replace existing review queue entries for processed books", value=True, ) ocr_btn = gr.Button("Run OCR →", elem_classes=["ss-btn-run"], interactive=True) ocr_log = gr.Textbox(label="Log", lines=10, interactive=False, elem_classes=["ss-log"]) ocr_copy_log_btn = gr.Button("Copy Log", size="sm") ocr_copy_log_btn.click( fn=None, inputs=[ocr_log], outputs=[], js="(logTxt) => { if (navigator && navigator.clipboard) { navigator.clipboard.writeText(logTxt || ''); } return []; }", ) ocr_run_event = ocr_btn.click( run_ocr, inputs=[ocr_page_selection, ocr_page_exclusion, ocr_replace_queue], outputs=[ocr_status, ocr_log, ocr_done], show_progress="minimal", trigger_mode="once", ) ocr_tab.select( _default_ocr_scope_values, outputs=[ocr_page_selection, ocr_page_exclusion], ) # Save scope AND push to OCR fields in one handler # Must be here — after ocr_page_selection is defined def _save_scope_and_push(book_id, inc, exc, title): status, table, log_txt = save_book_scope(book_id, inc, exc, title) return status, table, log_txt, (inc or "").strip(), (exc or "").strip() scope_save_btn.click( _save_scope_and_push, inputs=[scope_book_id, scope_include, scope_exclude, scope_title], outputs=[ingest_status, manifest_table, ingest_log, ocr_page_selection, ocr_page_exclusion], ) # ── STEP 4: REVIEW ──────────────────────────────────────────────── with gr.TabItem("③ Review", id="ss-review") as review_tab: gr.HTML("""

Review Workbench

Correct · Accept · Reject · Build gold training set

▶ Order: Click ↻ Load Review Queue   Review the page image and text   Edit text if needed · click ✓ Accept, ✎ Save Edit, or ✗ Reject   Use ← Prev / Next → to navigate without deciding
""") review_status = gr.HTML(_review_status_html()) training_feedback = gr.HTML() with gr.Row(): with gr.Column(scale=2): review_image = gr.Image( label="Page Render", type="filepath", height=420, ) item_info = gr.HTML() with gr.Column(scale=2): raw_text_box = gr.Textbox( label="Raw OCR", lines=6, interactive=False, ) final_text_box = gr.Textbox( label="Final Text (edit to correct)", lines=8, interactive=True, ) detected_font_display = gr.Textbox( label="Detected Font", interactive=False, scale=1, ) noise_input = gr.Textbox( label="Paste noise patterns to block (one per line — then Reject → NON_STORY_TEXT)", placeholder="e.g.\nWA alle\nCoS lic\ntd wht LN", lines=4, ) reason_code = gr.Dropdown( label="Reason code", choices=["","OCR_MISS","OCR_WRONG_WORD","DECORATIVE_FONT", "SPEECH_BUBBLE_ERROR","LOW_CONTRAST","SCAN_SKEW_BLUR", "NON_STORY_TEXT","LLM_OVER_CORRECTION","OTHER"], value="", ) conf_override = gr.Checkbox( label="Override confidence to 100% (page is perfect)", value=False, ) with gr.Row(): accept_btn = gr.Button("✓ Accept", elem_classes=["ss-btn-accept"]) edit_btn = gr.Button("✎ Save Edit", elem_classes=["ss-btn-edit"]) with gr.Row(): reject_btn = gr.Button("✗ Reject", elem_classes=["ss-btn-reject"]) quar_btn = gr.Button("⚑ Quarantine", elem_classes=["ss-btn-quar"]) with gr.Row(): prev_btn = gr.Button("← Prev", elem_classes=["ss-btn-next"]) next_btn = gr.Button("Next →", elem_classes=["ss-btn-next"]) load_review_btn = gr.Button("↻ Load Review Queue", elem_classes=["ss-btn-next"]) current_idx = gr.State(0) def load_review(): img, raw, info, done, total, font = get_review_item(0) status = _review_status_html() return 0, status, img, raw, raw, info, font def next_item(idx): new_idx = idx + 1 img, raw, info, done, total, font = get_review_item(new_idx) return new_idx, img, raw, raw, info, font def do_accept(idx, final, reviewer, reason, conf_ov=False): action = "edited" if final.strip() != "" else "accepted" result = save_review_decision( idx, final, action, reviewer, reason, conf_override=bool(conf_ov) ) fb = result[0] status = _review_status_html() new_idx = idx + 1 img2, raw2, info2, _, _, font2 = get_review_item(new_idx) return fb, status, new_idx, img2, raw2, raw2, info2, font2 def do_reject(idx, final, reviewer, reason): result = save_review_decision(idx, final, "rejected", reviewer, reason) fb = result[0] status = _review_status_html() new_idx = idx + 1 img2, raw2, info2, _, _, font2 = get_review_item(new_idx) return fb, status, new_idx, img2, raw2, raw2, info2, font2 def do_quarantine(idx, final, reviewer, reason): result = save_review_decision(idx, final, "quarantined", reviewer, reason) fb = result[0] status = _review_status_html() new_idx = idx + 1 img2, raw2, info2, _, _, font2 = get_review_item(new_idx) return fb, status, new_idx, img2, raw2, raw2, info2, font2 action_outputs = [training_feedback, review_status, current_idx, review_image, raw_text_box, final_text_box, item_info, detected_font_display] review_load_outputs = [current_idx, review_status, review_image, raw_text_box, final_text_box, item_info, detected_font_display] load_review_btn.click( load_review, outputs=review_load_outputs ) review_tab.select(load_review, outputs=review_load_outputs) ocr_done.change(load_review, outputs=review_load_outputs) pipeline_done.change(load_review, outputs=review_load_outputs) accept_btn.click(do_accept, inputs=[current_idx, final_text_box, noise_input, reason_code, conf_override], outputs=action_outputs) edit_btn.click(do_accept, inputs=[current_idx, final_text_box, noise_input, reason_code, conf_override], outputs=action_outputs) reject_btn.click(do_reject, inputs=[current_idx, final_text_box, noise_input, reason_code], outputs=action_outputs) quar_btn.click(do_quarantine, inputs=[current_idx, final_text_box, noise_input, reason_code], outputs=action_outputs) def _next(idx): new_idx = idx + 1 img, raw, info, _, _, font = get_review_item(new_idx) return new_idx, img, raw, raw, info, font def _prev(idx): new_idx = max(0, idx - 1) img, raw, info, _, _, font = get_review_item(new_idx) return new_idx, img, raw, raw, info, font prev_btn.click(_prev, inputs=[current_idx], outputs=[current_idx, review_image, raw_text_box, final_text_box, item_info, detected_font_display]) next_btn.click(_next, inputs=[current_idx], outputs=[current_idx, review_image, raw_text_box, final_text_box, item_info, detected_font_display]) # Review now auto-loads on tab select and after OCR run completion. # ── STEP 5: EXPORT ──────────────────────────────────────────────── with gr.TabItem("④ Export", id="ss-export") as export_tab: gr.HTML("""

Codex Export

Clean JSONL · Gold training set · Auto-feed Codex

""") export_status = gr.HTML(_export_status_html()) gr.HTML("""
What gets exported
codex_export_[batch].jsonl — all accepted/edited text with full provenance, ready for Codex fingerprint analysis.
gold_set_[batch].jsonl — your labelled corrections for future Surya fine-tuning. The more you correct, the better your next training run will be.
""") export_btn = gr.Button("Export to Codex →", elem_classes=["ss-btn-run"]) export_log = gr.Textbox(label="Export log", lines=6, interactive=False, elem_classes=["ss-log"]) with gr.Row(): codex_download = gr.File(label="Codex JSONL", interactive=False) gold_download = gr.File(label="Gold Training Set", interactive=False) export_btn.click( run_export, outputs=[export_status, export_log, codex_download, gold_download], ) ingest_tab.select(lambda: _wizard_header_html(1), outputs=[wizard_header]) ocr_tab.select(lambda: _wizard_header_html(3), outputs=[wizard_header]) review_tab.select(lambda: _wizard_header_html(4), outputs=[wizard_header]) export_tab.select(lambda: _wizard_header_html(5), outputs=[wizard_header])