Spaces:
Running
Running
| #!/usr/bin/env python3 | |
| """ | |
| Scan2Doc Pro v6.0 β Complete OCR-to-Word Converter | |
| Single file: Engine + GUI + Installer | |
| """ | |
| #!/usr/bin/env python3 | |
| """ | |
| Scan2Doc Pro v6.0 β Complete OCR-to-Word Application | |
| ====================================================== | |
| Complete rewrite with PDF support, speed optimizations, and enhanced processing. | |
| Major improvements over v5: | |
| - PDF Support: pymupdf (fitz) for PDF-to-image conversion, text & scanned PDFs | |
| - Speed: bilateralFilter replaces fastNlMeansDenoising, parallel PDF processing | |
| - Error Handling: try/except blocks, atexit temp cleanup, graceful degradation | |
| - Enhanced Image Processing: auto-rotate detection, contrast stretching, | |
| morphological cleanup, adaptive thresholding | |
| - Better Deskew: Hough line voting for more accurate angle detection | |
| - All v5 features retained: multi-PSM OCR, cell-by-cell table extraction, | |
| hOCR parsing, Persian/Arabic support, gridless table detection, | |
| paragraph reconstruction, heading detection, professional Word output | |
| Supports: Persian (Farsi), Arabic, and English | |
| Engine: Tesseract OCR + OpenCV + pymupdf + python-docx | |
| """ | |
| import atexit | |
| import argparse | |
| import math | |
| import os | |
| import re | |
| import subprocess | |
| import sys | |
| import tempfile | |
| import time | |
| import warnings | |
| from collections import defaultdict | |
| from concurrent.futures import ThreadPoolExecutor, as_completed | |
| from dataclasses import dataclass, field | |
| from pathlib import Path | |
| from typing import Any, Callable, Dict, List, Optional, Tuple | |
| import cv2 | |
| import numpy as np | |
| import pytesseract | |
| from PIL import Image | |
| from bs4 import BeautifulSoup | |
| from docx import Document | |
| from docx.shared import Inches, Pt, Cm | |
| from docx.enum.text import WD_ALIGN_PARAGRAPH | |
| from docx.enum.table import WD_TABLE_ALIGNMENT | |
| from docx.oxml.ns import qn, nsdecls | |
| from docx.oxml import parse_xml | |
| try: | |
| import fitz # pymupdf | |
| HAS_FITZ = True | |
| except ImportError: | |
| HAS_FITZ = False | |
| print("[WARN] pymupdf not installed β PDF support disabled. pip install pymupdf") | |
| warnings.filterwarnings("ignore", category=FutureWarning) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Temp file cleanup via atexit | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| _TEMP_FILES: list = [] | |
| def _cleanup_temp_files(): | |
| for fp in _TEMP_FILES: | |
| try: | |
| if os.path.exists(fp): | |
| os.unlink(fp) | |
| except Exception: | |
| pass | |
| _TEMP_FILES.clear() | |
| atexit.register(_cleanup_temp_files) | |
| def _register_temp(suffix=".png"): | |
| """Create a temp file and register for automatic cleanup.""" | |
| fd, path = tempfile.mkstemp(suffix=suffix, prefix="_s2d_v6_") | |
| os.close(fd) | |
| _TEMP_FILES.append(path) | |
| return path | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Constants β Persian/Arabic OCR character fixes (COMPLETE) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| PERSIAN_FIXES = { | |
| # Arabic β Persian letters | |
| 'Ω': 'Ϊ©', 'Ω': 'Ϋ', 'Ψ€': 'Ω', 'Ψ₯': 'Ψ§', 'Ψ£': 'Ψ§', 'Ω±': 'Ψ§', | |
| # Arabic β Persian digits | |
| 'Ω€': 'Ϋ΄', 'Ω₯': 'Ϋ΅', 'Ω¦': 'ΫΆ', 'Ω§': 'Ϋ·', 'Ω¨': 'ΫΈ', 'Ω©': 'ΫΉ', | |
| 'Ω ': 'Ϋ°', 'Ω‘': 'Ϋ±', 'Ω’': 'Ϋ²', 'Ω£': 'Ϋ³', | |
| # Arabic letter forms β Persian | |
| 'Ψ©': 'Ω', | |
| # Remove diacritics (tashkeel) | |
| 'Ω': '', 'Ω': '', 'Ω': '', 'Ω': '', 'Ω': '', 'Ω': '', | |
| # KEEP Persian punctuation (not convert to ASCII!) | |
| 'Ψ': 'Ψ', 'Ψ': 'Ψ', 'Ψ': 'Ψ', | |
| # Remove kashida | |
| 'Ω': '', | |
| # Unicode normalization | |
| '\u0640': '', # Tatweel | |
| '\u0670': '', # Superscript alef | |
| '\u200c': '', # Zero-width non-joiner | |
| '\u200d': '', # Zero-width joiner | |
| '\u00a0': ' ', # Non-breaking space | |
| '\u200b': '', # Zero-width space | |
| '\ufeff': '', # BOM | |
| } | |
| def fix_persian(text: str) -> str: | |
| """Complete Persian text fix: ArabicβPersian, numbers, diacritics, whitespace.""" | |
| if not text: | |
| return text | |
| result = [PERSIAN_FIXES.get(ch, ch) for ch in text] | |
| text = ''.join(result) | |
| text = re.sub(r'(?<=[Ϋ°-ΫΉ])\s*(?=[Ϋ°-ΫΉ])', '', text) | |
| text = re.sub(r'\s+', ' ', text) | |
| return text.strip() | |
| # Alias for backward compatibility | |
| fix_persian_text = fix_persian | |
| def detect_rtl(text: str) -> bool: | |
| """Detect if text is predominantly RTL (Persian/Arabic/Hebrew).""" | |
| if not text: | |
| return False | |
| rtl_chars = 0 | |
| ltr_chars = 0 | |
| for c in text: | |
| code = ord(c) | |
| if (0x0600 <= code <= 0x06FF or 0x0750 <= code <= 0x077F or | |
| 0x0590 <= code <= 0x05FF or 0xFB50 <= code <= 0xFDFF or | |
| 0xFE70 <= code <= 0xFEFF or 0x0590 <= code <= 0x08FF or | |
| 0xFB00 <= code <= 0xFBFF or 0xFE00 <= code <= 0xFEFF): | |
| rtl_chars += 1 | |
| elif 'a' <= c.lower() <= 'z': | |
| ltr_chars += 1 | |
| return rtl_chars > ltr_chars | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Data Classes | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class BBox: | |
| x: int; y: int; w: int; h: int | |
| def x2(self): return self.x + self.w | |
| def y2(self): return self.y + self.h | |
| def center_x(self): return self.x + self.w / 2 | |
| def center_y(self): return self.y + self.h / 2 | |
| def area(self): return self.w * self.h | |
| def overlaps(self, other, threshold=0.5): | |
| x1 = max(self.x, other.x); y1 = max(self.y, other.y) | |
| x2 = min(self.x2, other.x2); y2 = min(self.y2, other.y2) | |
| if x2 <= x1 or y2 <= y1: return False | |
| return (x2 - x1) * (y2 - y1) / max(self.area, 1) > threshold | |
| class TextBlock: | |
| bbox: BBox | |
| text: str = "" | |
| font_size: float = 12.0 | |
| is_bold: bool = False | |
| is_italic: bool = False | |
| is_rtl: bool = False | |
| confidence: float = 0.0 | |
| element_type: str = "paragraph" | |
| alignment: str = "left" | |
| column: int = 0 | |
| class TableData: | |
| bbox: BBox | |
| rows: int = 0 | |
| cols: int = 0 | |
| cells: List[List[str]] = field(default_factory=list) | |
| class LineElement: | |
| bbox: BBox | |
| orientation: str = "horizontal" | |
| is_table_border: bool = False | |
| thickness: int = 1 | |
| class ImageRegion: | |
| bbox: BBox | |
| aspect_ratio: float = 0.0 | |
| class PageStatistics: | |
| text_blocks: int = 0 | |
| tables: int = 0 | |
| lines: int = 0 | |
| image_regions: int = 0 | |
| columns: int = 1 | |
| overall_confidence: float = 0.0 | |
| processing_time: float = 0.0 | |
| is_rtl: bool = False | |
| total_words: int = 0 | |
| low_confidence_words: int = 0 | |
| class LayoutResult: | |
| text_blocks: List[TextBlock] = field(default_factory=list) | |
| tables: List[TableData] = field(default_factory=list) | |
| lines: List[LineElement] = field(default_factory=list) | |
| image_regions: List[ImageRegion] = field(default_factory=list) | |
| columns: int = 1 | |
| column_boundaries: List[Tuple[int, int]] = field(default_factory=list) | |
| is_rtl: bool = False | |
| median_font_size: float = 12.0 | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # ImageProcessor β v6: PDF support, faster denoise, auto-rotate, enhanced preprocessing | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class ImageProcessor: | |
| def __init__(self, dpi=300): | |
| self.dpi = dpi | |
| # ββ PDF to Images ββ | |
| def pdf_to_images(self, pdf_path, max_pages=None): | |
| """Convert PDF pages to images using pymupdf (fitz). | |
| Returns list of numpy arrays (BGR) for each page.""" | |
| if not HAS_FITZ: | |
| raise ImportError("pymupdf is required for PDF support: pip install pymupdf") | |
| try: | |
| doc = fitz.open(pdf_path) | |
| except Exception as e: | |
| print(f" [PDF] Error opening {pdf_path}: {e}") | |
| return [] | |
| pages = [] | |
| page_count = len(doc) | |
| if max_pages: | |
| page_count = min(page_count, max_pages) | |
| print(f" [PDF] {os.path.basename(pdf_path)}: {page_count} page(s)") | |
| for i in range(page_count): | |
| try: | |
| page = doc[i] | |
| # Render at target DPI | |
| zoom = self.dpi / 72.0 | |
| mat = fitz.Matrix(zoom, zoom) | |
| pix = page.get_pixmap(matrix=mat, alpha=False) | |
| # Convert to numpy array | |
| img_data = np.frombuffer(pix.samples, dtype=np.uint8) | |
| img = img_data.reshape(pix.height, pix.width, 3) | |
| # pymupdf gives RGB, OpenCV expects BGR | |
| img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) | |
| pages.append(img) | |
| print(f" [PDF] Page {i + 1}/{page_count}: {pix.width}x{pix.height}") | |
| except Exception as e: | |
| print(f" [PDF] Warning: Page {i + 1} failed: {e}") | |
| continue | |
| doc.close() | |
| return pages | |
| def is_scanned_pdf(self, pdf_path): | |
| """Check if PDF is scanned (image-based) or text-based. | |
| Returns (is_scanned, text_page_count, image_page_count).""" | |
| if not HAS_FITZ: | |
| return True, 0, 0 | |
| try: | |
| doc = fitz.open(pdf_path) | |
| text_pages = 0 | |
| image_pages = 0 | |
| for page in doc: | |
| text = page.get_text().strip() | |
| if len(text) > 50: | |
| text_pages += 1 | |
| else: | |
| image_pages += 1 | |
| doc.close() | |
| is_scanned = image_pages > text_pages | |
| return is_scanned, text_pages, image_pages | |
| except Exception: | |
| return True, 0, 0 | |
| # ββ Image Loading ββ | |
| def load_image(self, path): | |
| """Load image with proper error handling. Supports image files and PDFs.""" | |
| ext = Path(path).suffix.lower() | |
| if ext == '.pdf': | |
| pages = self.pdf_to_images(path) | |
| if not pages: | |
| raise FileNotFoundError(f"Cannot load PDF: {path}") | |
| return pages[0] # Return first page for single-page processing | |
| img = cv2.imread(path, cv2.IMREAD_COLOR) | |
| if img is None: | |
| raise FileNotFoundError(f"Cannot load image: {path}") | |
| print(f" [IMG] Loaded: {os.path.basename(path)} ({img.shape[1]}x{img.shape[0]})") | |
| return img | |
| # ββ Color Handling ββ | |
| def handle_color(self, img): | |
| """Better color image handling: detect if grayscale vs color and convert appropriately.""" | |
| if len(img.shape) == 2: | |
| return img # Already grayscale | |
| # Check if image is effectively grayscale (low color variance) | |
| hsv = cv2.cvtColor(img, cv2.COLOR_BGR2HSV) | |
| s_channel = hsv[:, :, 1] | |
| mean_saturation = np.mean(s_channel) | |
| if mean_saturation < 15: | |
| # Low saturation β treat as grayscale for better OCR | |
| gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) | |
| print(f" [IMG] Low saturation ({mean_saturation:.1f}), using grayscale") | |
| return gray | |
| # For color images, convert to grayscale but enhance contrast | |
| gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) | |
| return gray | |
| # ββ Denoising (v6: faster bilateralFilter) ββ | |
| def denoise(self, img, strength=10): | |
| """Fast denoising using bilateralFilter instead of slow fastNlMeansDenoising. | |
| bilateralFilter preserves edges while smoothing noise β much faster.""" | |
| if len(img.shape) == 3: | |
| gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray = img.copy() | |
| try: | |
| # bilateralFilter: d=9, sigmaColor=75, sigmaSpace=75 | |
| # Much faster than fastNlMeansDenoising with good edge preservation | |
| denoised = cv2.bilateralFilter(gray, d=9, sigmaColor=75, sigmaSpace=75) | |
| return denoised | |
| except Exception as e: | |
| print(f" [IMG] Denoise fallback: {e}") | |
| return gray | |
| def denoise_fast(self, img): | |
| """Ultra-fast denoising for time-critical paths (e.g., parallel processing).""" | |
| if len(img.shape) == 3: | |
| gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray = img.copy() | |
| return cv2.bilateralFilter(gray, d=5, sigmaColor=50, sigmaSpace=50) | |
| def denoise_aggressive(self, img): | |
| """Aggressive denoising for very noisy images.""" | |
| if len(img.shape) == 3: | |
| gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray = img.copy() | |
| # Step 1: Bilateral filter (edge-preserving) | |
| filtered = cv2.bilateralFilter(gray, d=9, sigmaColor=75, sigmaSpace=75) | |
| # Step 2: Morphological opening to remove small noise | |
| kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (2, 2)) | |
| opened = cv2.morphologyEx(filtered, cv2.MORPH_OPEN, kernel) | |
| # Step 3: Median blur for salt-and-pepper noise | |
| denoised = cv2.medianBlur(opened, 3) | |
| return denoised | |
| def detect_noise_level(self, img): | |
| """Detect noise level in image (0-100).""" | |
| if len(img.shape) == 3: | |
| gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray = img.copy() | |
| # Use Laplacian variance as noise estimate | |
| laplacian = cv2.Laplacian(gray, cv2.CV_64F) | |
| noise_level = np.var(laplacian) | |
| # Normalize to 0-100 scale | |
| return min(100, noise_level / 100) | |
| # ββ Contrast Enhancement ββ | |
| def enhance_contrast(self, img): | |
| """Apply contrast stretching for better OCR preprocessing.""" | |
| if len(img.shape) == 3: | |
| gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray = img.copy() | |
| # Check if image is already good contrast | |
| p_low, p_high = np.percentile(gray, [2, 98]) | |
| if p_high - p_low > 100: | |
| return gray # Already good contrast | |
| # Multi-stage contrast enhancement | |
| # Stage 1: CLAHE for local contrast | |
| clahe = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8, 8)) | |
| enhanced = clahe.apply(gray) | |
| # Stage 2: Contrast stretching | |
| p_low, p_high = np.percentile(enhanced, [1, 99]) | |
| if p_high - p_low > 10: | |
| enhanced = np.clip((enhanced.astype(np.float32) - p_low) / (p_high - p_low) * 255, 0, 255) | |
| enhanced = enhanced.astype(np.uint8) | |
| # Stage 3: Sharpening for very blurry images | |
| variance = np.var(cv2.Laplacian(enhanced, cv2.CV_64F)) | |
| if variance < 100: # Very blurry | |
| kernel = np.array([[-1,-1,-1], [-1,9,-1], [-1,-1,-1]]) | |
| enhanced = cv2.filter2D(enhanced, -1, kernel) | |
| enhanced = np.clip(enhanced, 0, 255).astype(np.uint8) | |
| return enhanced | |
| # ββ Auto-rotate Detection ββ | |
| def detect_rotation(self, img): | |
| """Detect text orientation and suggest rotation angle. | |
| Uses minAreaRect on text pixels for dominant orientation.""" | |
| if len(img.shape) == 3: | |
| gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray = img.copy() | |
| # Use adaptive thresholding for better text detection | |
| try: | |
| binary = cv2.adaptiveThreshold( | |
| gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, | |
| cv2.THRESH_BINARY_INV, 31, 10) | |
| except Exception: | |
| _, binary = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU) | |
| # Dilate to connect text into regions | |
| kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (5, 5)) | |
| dilated = cv2.dilate(binary, kernel, iterations=2) | |
| coords = np.column_stack(np.where(dilated > 0)) | |
| if len(coords) < 100: | |
| return 0.0 | |
| # Use minAreaRect to find dominant orientation | |
| angle = cv2.minAreaRect(coords)[-1] | |
| if angle < -45: | |
| angle = 90 + angle | |
| # Only return significant angles | |
| if abs(angle) < 0.5: | |
| return 0.0 | |
| # Clamp to reasonable range | |
| return max(-15.0, min(15.0, angle)) | |
| # ββ Improved Deskew ββ | |
| def deskew(self, img): | |
| """Improved deskew using Hough line voting for more accurate angle detection. | |
| Falls back to minAreaRect if Hough detection fails.""" | |
| if len(img.shape) == 3: | |
| gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray = img.copy() | |
| # Method 1: Hough line voting (more accurate for document images) | |
| edges = cv2.Canny(gray, 50, 150, apertureSize=3) | |
| lines = cv2.HoughLinesP(edges, 1, np.pi / 180, 100, | |
| minLineLength=gray.shape[1] // 4, maxLineGap=10) | |
| if lines is not None and len(lines) > 5: | |
| angles = [] | |
| for line in lines: | |
| pts = line.reshape(4) | |
| x1, y1, x2, y2 = int(pts[0]), int(pts[1]), int(pts[2]), int(pts[3]) | |
| if abs(x2 - x1) > abs(y2 - y1): # Mostly horizontal lines | |
| angle = math.degrees(math.atan2(y2 - y1, x2 - x1)) | |
| if abs(angle) < 45: # Consider wider range of angles | |
| angles.append(angle) | |
| if angles: | |
| # Use median angle (robust to outliers) | |
| angle = np.median(angles) | |
| if abs(angle) < 0.3: | |
| return img, 0.0 | |
| h, w = img.shape[:2] | |
| M = cv2.getRotationMatrix2D((w // 2, h // 2), angle, 1.0) | |
| # Calculate new bounding box to avoid cropping | |
| cos_a = abs(M[0, 0]) | |
| sin_a = abs(M[0, 1]) | |
| new_w = int(h * sin_a + w * cos_a) | |
| new_h = int(h * cos_a + w * sin_a) | |
| M[0, 2] += (new_w - w) / 2 | |
| M[1, 2] += (new_h - h) / 2 | |
| rotated = cv2.warpAffine(img, M, (new_w, new_h), flags=cv2.INTER_CUBIC, | |
| borderMode=cv2.BORDER_REPLICATE) | |
| print(f" [IMG] Deskewed by {angle:.2f}Β° (Hough)") | |
| return rotated, angle | |
| # Method 2: Fallback to minAreaRect | |
| inverted = cv2.bitwise_not(gray) | |
| coords = np.column_stack(np.where(inverted > 0)) | |
| if len(coords) < 50: | |
| return img, 0.0 | |
| angle = cv2.minAreaRect(coords)[-1] | |
| if angle < -45: | |
| angle = 90 + angle | |
| if abs(angle) < 0.3: | |
| return img, 0.0 | |
| h, w = img.shape[:2] | |
| M = cv2.getRotationMatrix2D((w // 2, h // 2), angle, 1.0) | |
| rotated = cv2.warpAffine(img, M, (w, h), flags=cv2.INTER_CUBIC, | |
| borderMode=cv2.BORDER_REPLICATE) | |
| print(f" [IMG] Deskewed by {angle:.2f}Β° (minAreaRect)") | |
| return rotated, angle | |
| # ββ Binarization (v6: multiple methods) ββ | |
| def binarize(self, img, method="adaptive"): | |
| """Binarize image using specified method. | |
| Methods: 'adaptive' (default), 'otsu', 'sauvola'""" | |
| if len(img.shape) == 3: | |
| gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray = img.copy() | |
| if method == "otsu": | |
| _, binary = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU) | |
| return binary | |
| elif method == "sauvola": | |
| # Sauvola-like adaptive threshold (simulated with Gaussian adaptive) | |
| return cv2.adaptiveThreshold( | |
| gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, | |
| cv2.THRESH_BINARY, 31, 12) | |
| else: # adaptive | |
| return cv2.adaptiveThreshold( | |
| gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, | |
| cv2.THRESH_BINARY, 31, 10) | |
| # ββ Morphological Cleanup ββ | |
| def morphological_cleanup(self, binary): | |
| """Clean up binarized image with morphological operations. | |
| Remove small noise dots and fill small holes.""" | |
| if len(binary.shape) == 3: | |
| gray = cv2.cvtColor(binary, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray = binary.copy() | |
| # Remove small noise (opening) | |
| kernel_small = cv2.getStructuringElement(cv2.MORPH_RECT, (2, 2)) | |
| cleaned = cv2.morphologyEx(gray, cv2.MORPH_OPEN, kernel_small) | |
| # Fill small holes (closing) | |
| kernel_medium = cv2.getStructuringElement(cv2.MORPH_RECT, (3, 3)) | |
| cleaned = cv2.morphologyEx(cleaned, cv2.MORPH_CLOSE, kernel_medium) | |
| return cleaned | |
| # ββ Main Preparation Pipeline ββ | |
| def prepare_for_ocr(self, path): | |
| """Full preprocessing pipeline for file input.""" | |
| img = self.load_image(path) | |
| gray = self.handle_color(img) | |
| enhanced = self.enhance_contrast(gray) | |
| # Smart denoising based on noise level | |
| noise_level = self.detect_noise_level(enhanced) | |
| if noise_level > 50: | |
| print(f" [IMG] High noise detected ({noise_level:.0f}), using aggressive denoising") | |
| denoised = self.denoise_aggressive(enhanced) | |
| else: | |
| denoised = self.denoise(enhanced) | |
| deskewed, angle = self.deskew(denoised) | |
| binary = self.binarize(deskewed) | |
| binary = self.morphological_cleanup(binary) | |
| return deskewed, binary | |
| def prepare_for_ocr_from_array(self, img): | |
| """Preprocessing pipeline for numpy array input (e.g., PDF pages). | |
| Returns (gray, binary) for downstream processing.""" | |
| gray = self.handle_color(img) | |
| enhanced = self.enhance_contrast(gray) | |
| # Smart denoising based on noise level | |
| noise_level = self.detect_noise_level(enhanced) | |
| if noise_level > 50: | |
| denoised = self.denoise_aggressive(enhanced) | |
| else: | |
| denoised = self.denoise(enhanced) | |
| deskewed, angle = self.deskew(denoised) | |
| binary = self.binarize(deskewed) | |
| binary = self.morphological_cleanup(binary) | |
| return deskewed, binary | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # LayoutAnalyzer β v6: Enhanced with better error handling | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class LayoutAnalyzer: | |
| def __init__(self, min_block_area=100): | |
| self.min_block_area = min_block_area | |
| # ββ Multi-Scale Line Detection ββ | |
| def detect_lines_multiscale(self, binary, min_length_ratio=0.10): | |
| """Detect horizontal and vertical lines at multiple scales. | |
| Enhanced with error handling for edge cases.""" | |
| h, w = binary.shape | |
| if h < 10 or w < 10: | |
| return [], [] | |
| inverted = ~binary if len(binary.shape) == 2 else ~cv2.cvtColor(binary, cv2.COLOR_BGR2GRAY) | |
| all_h, all_v = [], [] | |
| for scale in [8, 10, 15, 20, 30, 40]: | |
| try: | |
| # Horizontal lines | |
| kw = max(w // scale, 8) | |
| kh_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (kw, 1)) | |
| h_mask = cv2.morphologyEx(inverted, cv2.MORPH_OPEN, kh_kernel) | |
| contours, _ = cv2.findContours(h_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) | |
| for c in contours: | |
| x, y, cw, ch = cv2.boundingRect(c) | |
| if cw > w * min_length_ratio: | |
| all_h.append(LineElement(BBox(x, y, cw, max(ch, 2)), "horizontal", thickness=max(ch, 1))) | |
| # Vertical lines | |
| kv = cv2.getStructuringElement(cv2.MORPH_RECT, (1, max(h // scale, 8))) | |
| v_mask = cv2.morphologyEx(inverted, cv2.MORPH_OPEN, kv) | |
| contours, _ = cv2.findContours(v_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) | |
| for c in contours: | |
| x, y, cw, ch = cv2.boundingRect(c) | |
| if ch > h * min_length_ratio: | |
| all_v.append(LineElement(BBox(x, y, max(cw, 2), ch), "vertical", thickness=max(cw, 1))) | |
| except Exception as e: | |
| continue # Skip failed scale gracefully | |
| # Hough lines for additional precision | |
| try: | |
| gray = binary if len(binary.shape) == 2 else cv2.cvtColor(binary, cv2.COLOR_BGR2GRAY) | |
| edges = cv2.Canny(gray, 50, 150, apertureSize=3) | |
| min_len = max(w * 0.20, 80) | |
| lines = cv2.HoughLinesP(edges, 1, np.pi / 180, 150, | |
| minLineLength=min_len, maxLineGap=10) | |
| if lines is not None: | |
| for line in lines: | |
| x1, y1, x2, y2 = line[:4] | |
| dx, dy = abs(x2 - x1), abs(y2 - y1) | |
| length = math.sqrt(dx * dx + dy * dy) | |
| if length < min_len * 0.4: | |
| continue | |
| if dy < 5 and dx > dy * 3: | |
| all_h.append(LineElement(BBox(min(x1, x2), min(y1, y2), | |
| abs(x2 - x1), max(abs(y2 - y1), 2)), | |
| "horizontal", thickness=2)) | |
| elif dx < 5 and dy > dx * 3: | |
| all_v.append(LineElement(BBox(min(x1, x2), min(y1, y2), | |
| max(abs(x2 - x1), 2), abs(y2 - y1)), | |
| "vertical", thickness=2)) | |
| except Exception: | |
| pass | |
| # Detect underlines | |
| self._detect_underlines(inverted, w, h, all_h) | |
| all_h = self._merge_lines(all_h, "horizontal") | |
| all_v = self._merge_lines(all_v, "vertical") | |
| return all_h, all_v | |
| def _detect_underlines(self, inverted, w, h, all_lines): | |
| """Detect short horizontal underlines (typically under headings).""" | |
| for kw_ratio in [4, 5, 6]: | |
| try: | |
| kw = max(w // kw_ratio, 30) | |
| kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (kw, 1)) | |
| mask = cv2.morphologyEx(inverted, cv2.MORPH_OPEN, kernel) | |
| contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) | |
| for c in contours: | |
| x, y, cw, ch = cv2.boundingRect(c) | |
| if w * 0.05 < cw < w * 0.40 and ch < h * 0.005: | |
| all_lines.append(LineElement(BBox(x, y, cw, max(ch, 2)), | |
| "horizontal", thickness=max(ch, 1))) | |
| except Exception: | |
| continue | |
| def _merge_lines(self, lines, orientation): | |
| if not lines: | |
| return [] | |
| key = (lambda l: l.bbox.y) if orientation == "horizontal" else (lambda l: l.bbox.x) | |
| lines.sort(key=key) | |
| merged = [] | |
| for line in lines: | |
| if merged: | |
| last = merged[-1] | |
| close = abs(key(line) - key(last)) < 8 | |
| if orientation == "horizontal": | |
| close = close and abs(line.bbox.center_x - last.bbox.center_x) < last.bbox.w * 0.5 | |
| else: | |
| close = close and abs(line.bbox.center_y - last.bbox.center_y) < last.bbox.h * 0.5 | |
| if close: | |
| x1 = min(last.bbox.x, line.bbox.x) | |
| y1 = min(last.bbox.y, line.bbox.y) | |
| x2 = max(last.bbox.x2, line.bbox.x2) | |
| y2 = max(last.bbox.y2, line.bbox.y2) | |
| merged[-1] = LineElement(BBox(x1, y1, x2 - x1, y2 - y1), orientation, | |
| thickness=max(last.thickness, line.thickness)) | |
| continue | |
| merged.append(line) | |
| return merged | |
| # ββ Table Detection ββ | |
| def detect_tables(self, binary, h_lines, v_lines): | |
| h, w = binary.shape | |
| tables = [] | |
| long_h = [l for l in h_lines if l.bbox.w > w * 0.15] | |
| long_v = [l for l in v_lines if l.bbox.h > h * 0.10] | |
| if len(long_h) < 2 or len(long_v) < 1: | |
| return tables | |
| combined = np.zeros((h, w), dtype=np.uint8) | |
| for line in long_h: | |
| cv2.line(combined, (line.bbox.x, line.bbox.y), (line.bbox.x2, line.bbox.y2), | |
| 255, max(line.thickness, 1)) | |
| for line in long_v: | |
| cv2.line(combined, (line.bbox.x, line.bbox.y), (line.bbox.x2, line.bbox.y2), | |
| 255, max(line.thickness, 1)) | |
| kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (5, 5)) | |
| dilated = cv2.dilate(combined, kernel, iterations=2) | |
| num_labels, labels, stats, _ = cv2.connectedComponentsWithStats(dilated, 8) | |
| for i in range(1, num_labels): | |
| x, y = stats[i, cv2.CC_STAT_LEFT], stats[i, cv2.CC_STAT_TOP] | |
| bw, bh = stats[i, cv2.CC_STAT_WIDTH], stats[i, cv2.CC_STAT_HEIGHT] | |
| area = stats[i, cv2.CC_STAT_AREA] | |
| if bw < w * 0.10 or bh < h * 0.02 or area < 400: | |
| continue | |
| h_in = sum(1 for l in long_h if l.bbox.y >= y - 5 and l.bbox.y <= y + bh + 5) | |
| v_in = sum(1 for l in long_v if l.bbox.x >= x - 5 and l.bbox.x <= x + bw + 5) | |
| if h_in < 2 or v_in < 1: | |
| continue | |
| t_hl = [l for l in long_h if l.bbox.y >= y - 5 and l.bbox.y <= y + bh + 5] | |
| t_vl = [l for l in long_v if l.bbox.x >= x - 5 and l.bbox.x <= x + bw + 5] | |
| table = TableData(bbox=BBox(x, y, bw, bh)) | |
| table.rows = max(len(t_hl) - 1, 1) | |
| table.cols = max(len(t_vl) - 1, 1) | |
| tables.append(table) | |
| print(f" [LAYOUT] Table: {x},{y} {bw}x{bh} ({table.rows}x{table.cols})") | |
| if len(tables) > 1: | |
| tables.sort(key=lambda t: t.bbox.area, reverse=True) | |
| kept = [] | |
| for t in tables: | |
| if not any(t.bbox.overlaps(e.bbox, 0.3) for e in kept): | |
| kept.append(t) | |
| tables = kept | |
| return tables | |
| # ββ Gridless Table Detection ββ | |
| def detect_gridless_tables(self, binary, text_blocks, existing_tables): | |
| """Detect tables WITHOUT grid lines β just aligned text in rows/columns.""" | |
| h, w = binary.shape | |
| if not text_blocks or len(text_blocks) < 4: | |
| return [] | |
| existing_bboxes = [t.bbox for t in existing_tables] | |
| sorted_blocks = sorted(text_blocks, key=lambda b: (b.bbox.y, -b.bbox.x if b.is_rtl else b.bbox.x)) | |
| rows = [] | |
| current_row = [sorted_blocks[0]] | |
| for block in sorted_blocks[1:]: | |
| prev = current_row[-1] | |
| prev_cy = prev.bbox.center_y | |
| block_cy = block.bbox.center_y | |
| row_height = max(prev.bbox.h, block.bbox.h) | |
| if abs(prev_cy - block_cy) < row_height * 0.6: | |
| current_row.append(block) | |
| else: | |
| rows.append(current_row) | |
| current_row = [block] | |
| rows.append(current_row) | |
| if len(rows) < 3: | |
| return [] | |
| tables = [] | |
| i = 0 | |
| while i < len(rows) - 2: | |
| group = [rows[i]] | |
| row_xs = [b.bbox.x for b in sorted(rows[i], key=lambda b: b.bbox.x)] | |
| for j in range(i + 1, min(i + 8, len(rows))): | |
| next_xs = [b.bbox.x for b in sorted(rows[j], key=lambda b: b.bbox.x)] | |
| if len(next_xs) >= 2 and len(row_xs) >= 2: | |
| if len(next_xs) == len(row_xs): | |
| mismatch = sum(1 for a, b in zip(row_xs, next_xs) | |
| if abs(a - b) > w * 0.10) | |
| if mismatch <= len(row_xs) * 0.3: | |
| group.append(rows[j]) | |
| elif len(next_xs) == len(row_xs): | |
| group.append(rows[j]) | |
| if len(group) >= 3: | |
| all_blocks = [b for row in group for b in row] | |
| min_x = min(b.bbox.x for b in all_blocks) | |
| min_y = min(b.bbox.y for b in all_blocks) | |
| max_x2 = max(b.bbox.x2 for b in all_blocks) | |
| max_y2 = max(b.bbox.y2 for b in all_blocks) | |
| bbox = BBox(min_x, min_y, max_x2 - min_x, max_y2 - min_y) | |
| if any(bbox.overlaps(t, 0.3) for t in existing_bboxes): | |
| i += len(group) | |
| continue | |
| num_cols = max(len(row) for row in group) | |
| table = TableData(bbox=bbox, rows=len(group), cols=num_cols) | |
| tables.append(table) | |
| print(f" [LAYOUT] Gridless table: {bbox.x},{bbox.y} ({len(group)}x{num_cols})") | |
| i += len(group) | |
| else: | |
| i += 1 | |
| return tables | |
| # ββ Text Block Detection ββ | |
| def detect_text_blocks(self, binary): | |
| """Detect text blocks with enhanced splitting based on line gaps.""" | |
| h, w = binary.shape | |
| if h < 10 or w < 10: | |
| return [] | |
| try: | |
| tk = cv2.getStructuringElement(cv2.MORPH_RECT, (30, 5)) | |
| if len(binary.shape) == 3: | |
| gray_bin = cv2.cvtColor(binary, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray_bin = binary | |
| tm = cv2.morphologyEx(~gray_bin, cv2.MORPH_CLOSE, tk) | |
| tm = cv2.dilate(tm, cv2.getStructuringElement(cv2.MORPH_RECT, (20, 10)), iterations=2) | |
| except Exception as e: | |
| print(f" [LAYOUT] Block detection error: {e}") | |
| return [] | |
| blocks = [] | |
| for c in cv2.findContours(tm, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)[0]: | |
| x, y, bw, bh = cv2.boundingRect(c) | |
| if bw * bh < self.min_block_area or bw < w * 0.03 or bh < 8: | |
| continue | |
| blocks.append(BBox(x, y, bw, bh)) | |
| # Split large blocks based on internal line gaps | |
| split_blocks = [] | |
| for block in blocks: | |
| if block.h < 30: | |
| split_blocks.append(block) | |
| continue | |
| try: | |
| region = gray_bin[block.y:block.y2, block.x:block.x2] | |
| inverted = ~region | |
| h_proj = np.sum(inverted, axis=1).astype(np.float64) | |
| kernel_size = max(3, int(block.h * 0.03)) | |
| if kernel_size % 2 == 0: kernel_size += 1 | |
| if kernel_size > 1: | |
| h_proj_s = cv2.GaussianBlur(h_proj.reshape(1, -1), (1, kernel_size), 0).flatten() | |
| else: | |
| h_proj_s = h_proj | |
| threshold = np.max(h_proj_s) * 0.05 if np.max(h_proj_s) > 0 else 0 | |
| gap_start = None | |
| gaps = [] | |
| for y_pos in range(len(h_proj_s)): | |
| if h_proj_s[y_pos] <= threshold: | |
| if gap_start is None: | |
| gap_start = y_pos | |
| else: | |
| if gap_start is not None: | |
| gap_len = y_pos - gap_start | |
| if gap_len > block.h * 0.15: | |
| gaps.append((gap_start, y_pos)) | |
| gap_start = None | |
| if gaps and len(gaps) >= 1: | |
| prev_y = 0 | |
| for gap_start, gap_end in gaps: | |
| if gap_start - prev_y > 8: | |
| split_blocks.append(BBox( | |
| block.x, block.y + prev_y, | |
| block.w, gap_start - prev_y)) | |
| prev_y = gap_end | |
| if block.y2 - (block.y + prev_y) > 8: | |
| split_blocks.append(BBox( | |
| block.x, block.y + prev_y, | |
| block.w, block.y2 - (block.y + prev_y))) | |
| else: | |
| split_blocks.append(block) | |
| except Exception: | |
| split_blocks.append(block) | |
| split_blocks.sort(key=lambda b: (b.y, b.x)) | |
| return split_blocks | |
| # ββ Column Detection ββ | |
| def detect_columns(self, binary, text_blocks=None): | |
| """Detect columns using vertical projection + horizontal word distribution.""" | |
| h, w = binary.shape | |
| if h < 10 or w < 10: | |
| return 1, [(0, w)] | |
| if len(binary.shape) == 3: | |
| gray_bin = cv2.cvtColor(binary, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray_bin = binary | |
| inverted = ~gray_bin | |
| h_proj = np.sum(inverted, axis=0).astype(np.float64) | |
| ks = max(w // 100, 5) | |
| if ks % 2 == 0: ks += 1 | |
| h_proj_s = cv2.GaussianBlur(h_proj.reshape(1, -1), (ks, 1), 0).flatten() | |
| threshold = np.max(h_proj_s) * 0.02 | |
| in_text = h_proj_s > threshold | |
| gaps = [] | |
| gs = 0 | |
| in_gap = False | |
| for x in range(w): | |
| if not in_text[x]: | |
| if not in_gap: gs = x; in_gap = True | |
| else: | |
| if in_gap and (x - gs) > w * 0.05: | |
| gaps.append((gs, x)) | |
| in_gap = False | |
| if in_gap and (w - gs) > w * 0.05: | |
| gaps.append((gs, w)) | |
| valid = [] | |
| margin = w * 0.08 | |
| for gs_, ge in gaps: | |
| if gs_ < margin or ge > w - margin: | |
| continue | |
| gap_col = inverted[:, gs_:ge] | |
| density = np.sum(gap_col > 0) / (h * max(ge - gs_, 1)) | |
| if density < 0.03: | |
| valid.append((gs_, ge)) | |
| merged = [] | |
| for g in valid: | |
| if merged and g[0] - merged[-1][1] < w * 0.02: | |
| merged[-1] = (merged[-1][0], g[1]) | |
| else: | |
| merged.append(g) | |
| num_cols = 1 + len(merged) | |
| boundaries = [] | |
| if merged: | |
| boundaries.append((0, merged[0][0])) | |
| for i in range(len(merged) - 1): | |
| boundaries.append((merged[i][1], merged[i + 1][0])) | |
| boundaries.append((merged[-1][1], w)) | |
| else: | |
| boundaries = [(0, w)] | |
| # Horizontal word distribution analysis | |
| if num_cols == 1 and text_blocks and len(text_blocks) > 3: | |
| num_cols, boundaries = self._analyze_word_distribution( | |
| gray_bin, text_blocks, w, h, boundaries) | |
| if num_cols > 1: | |
| print(f" [LAYOUT] {num_cols} columns: {[(b[0], b[1]) for b in boundaries]}") | |
| return num_cols, boundaries | |
| def _analyze_word_distribution(self, binary, text_blocks, w, h, current_boundaries): | |
| """Analyze horizontal word distribution to detect columns.""" | |
| text_mask = np.zeros((h, w), dtype=np.uint8) | |
| for block in text_blocks: | |
| bx, by = max(0, block.x if hasattr(block, "x") else block.bbox.x), \ | |
| max(0, block.y if hasattr(block, "y") else block.bbox.y) | |
| bx2, by2 = min(w, block.x2 if hasattr(block, "x2") else block.bbox.x2), \ | |
| min(h, block.y2 if hasattr(block, "y2") else block.bbox.y2) | |
| text_mask[by:by2, bx:bx2] = 1 | |
| v_proj = np.sum(text_mask, axis=0).astype(np.float64) | |
| if np.max(v_proj) == 0: | |
| return 1, current_boundaries | |
| ks = max(w // 60, 7) | |
| if ks % 2 == 0: ks += 1 | |
| v_proj_smooth = cv2.GaussianBlur(v_proj.reshape(1, -1), (ks, 1), 0).flatten() | |
| v_proj_norm = v_proj_smooth / np.max(v_proj_smooth) | |
| gap_threshold = 0.05 | |
| in_gap = False | |
| gap_start = 0 | |
| found_gaps = [] | |
| for x in range(w): | |
| if v_proj_norm[x] < gap_threshold: | |
| if not in_gap: | |
| gap_start = x | |
| in_gap = True | |
| else: | |
| if in_gap: | |
| gap_width = x - gap_start | |
| if gap_width > w * 0.03: | |
| gap_region = text_mask[:, gap_start:x] | |
| vertical_coverage = np.sum(np.any(gap_region > 0, axis=1)) / h | |
| if vertical_coverage < 0.3: | |
| found_gaps.append((gap_start, x)) | |
| in_gap = False | |
| if found_gaps: | |
| merged_gaps = [found_gaps[0]] | |
| for g in found_gaps[1:]: | |
| if g[0] - merged_gaps[-1][1] < w * 0.01: | |
| merged_gaps[-1] = (merged_gaps[-1][0], g[1]) | |
| else: | |
| merged_gaps.append(g) | |
| if len(merged_gaps) >= 1: | |
| num_cols = len(merged_gaps) + 1 | |
| boundaries = [(0, merged_gaps[0][0])] | |
| for i in range(len(merged_gaps) - 1): | |
| boundaries.append((merged_gaps[i][1], merged_gaps[i + 1][0])) | |
| boundaries.append((merged_gaps[-1][1], w)) | |
| return num_cols, boundaries | |
| return 1, current_boundaries | |
| # ββ Image Region Detection ββ | |
| def detect_image_regions(self, binary): | |
| h, w = binary.shape | |
| if h < 10 or w < 10: | |
| return [] | |
| if len(binary.shape) == 3: | |
| gray = cv2.cvtColor(binary, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray = binary.copy() | |
| try: | |
| gx = cv2.Sobel(gray, cv2.CV_64F, 1, 0, ksize=3) | |
| gy = cv2.Sobel(gray, cv2.CV_64F, 0, 1, ksize=3) | |
| gm = np.sqrt(gx ** 2 + gy ** 2) | |
| gt = np.percentile(gm, 80) | |
| hg = (gm > gt).astype(np.uint8) * 255 | |
| hg = cv2.dilate(hg, cv2.getStructuringElement(cv2.MORPH_RECT, (15, 15)), iterations=2) | |
| except Exception: | |
| return [] | |
| regions = [] | |
| for c in cv2.findContours(hg, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)[0]: | |
| x, y, bw, bh = cv2.boundingRect(c) | |
| if bw * bh < 2000 or bw < w * 0.05 or bh < h * 0.03: | |
| continue | |
| regions.append(ImageRegion(BBox(x, y, bw, bh), bw / max(bh, 1))) | |
| return regions | |
| # ββ Main Pipeline ββ | |
| def analyze(self, binary, original): | |
| result = LayoutResult() | |
| h, w = binary.shape | |
| h_lines, v_lines = self.detect_lines_multiscale(binary) | |
| result.lines = h_lines + v_lines | |
| tables = self.detect_tables(binary, h_lines, v_lines) | |
| result.tables = tables | |
| for t in tables: | |
| for l in result.lines: | |
| if t.bbox.overlaps(l.bbox, 0.1): | |
| l.is_table_border = True | |
| tb = self.detect_text_blocks(binary) | |
| tb = [b for b in tb if not any(b.overlaps(t.bbox, 0.3) for t in tables)] | |
| img_regions = self.detect_image_regions(binary) | |
| result.image_regions = img_regions | |
| tb = [b for b in tb if not any(b.overlaps(ir.bbox, 0.3) for ir in img_regions)] | |
| nc, bnd = self.detect_columns(binary, tb) | |
| result.columns = nc | |
| result.column_boundaries = bnd | |
| # Gridless table detection | |
| tb_wrapped = [TextBlock(bbox=b) if isinstance(b, BBox) else b for b in tb] | |
| gridless = self.detect_gridless_tables(binary, tb_wrapped, tables) | |
| result.tables.extend(gridless) | |
| for b in tb: | |
| result.text_blocks.append(TextBlock(bbox=b)) | |
| result.is_rtl = self._detect_rtl(original) | |
| print(f" [LAYOUT] {len(result.text_blocks)} blocks, {len(result.tables)} tables, " | |
| f"{len(result.lines)} lines, {len(result.image_regions)} images, {nc} cols") | |
| return result | |
| def _detect_rtl(self, img): | |
| if len(img.shape) == 3: | |
| gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) | |
| else: | |
| gray = img.copy() | |
| _, b = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU) | |
| h, w = b.shape | |
| return np.sum(b[:, w // 2:] > 0) > np.sum(b[:, :w // 2] > 0) * 1.1 | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # TableExtractor β v6: Enhanced with error handling and contrast options | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class TableExtractor: | |
| def __init__(self, ocr_lang="fas+ara+eng"): | |
| self.ocr_lang = ocr_lang | |
| def extract_table(self, gray, table, ocr_engine): | |
| """Extract table content with cell-by-cell OCR, upscaling, and CLAHE.""" | |
| h, w = gray.shape | |
| x1 = max(0, table.bbox.x); y1 = max(0, table.bbox.y) | |
| x2 = min(w, table.bbox.x2); y2 = min(h, table.bbox.y2) | |
| table_img = gray[y1:y2, x1:x2] | |
| if table_img.size == 0: | |
| return TableData(bbox=table.bbox) | |
| try: | |
| _, binary = cv2.threshold(table_img, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU) | |
| except Exception: | |
| return TableData(bbox=table.bbox) | |
| th, tw = table_img.shape | |
| hp = self._find_positions(binary, axis=1, length=tw) | |
| vp = self._find_positions(binary, axis=0, length=th) | |
| if len(hp) < 2 or len(vp) < 2: | |
| text = ocr_engine.ocr_region(table_img, self.ocr_lang) | |
| return TableData(bbox=table.bbox, rows=1, cols=1, cells=[[fix_persian_text(text)]]) | |
| rows, cols = len(hp) - 1, len(vp) - 1 | |
| if rows > 50 or cols > 20: | |
| text = ocr_engine.ocr_region(table_img, self.ocr_lang) | |
| return TableData(bbox=table.bbox, rows=1, cols=1, cells=[[fix_persian_text(text)]]) | |
| cells = [] | |
| for r in range(rows): | |
| rc = [] | |
| for c in range(cols): | |
| cy1, cy2 = hp[r] + 3, hp[r + 1] - 3 | |
| cx1, cx2 = vp[c] + 3, vp[c + 1] - 3 | |
| if cy2 <= cy1 or cx2 <= cx1: | |
| rc.append(""); continue | |
| cy1, cy2 = max(0, cy1), min(th, cy2) | |
| cx1, cx2 = max(0, cx1), min(tw, cx2) | |
| cell = table_img[cy1:cy2, cx1:cx2] | |
| if cell.size == 0: | |
| rc.append(""); continue | |
| # Skip empty cells | |
| dark_pixels = np.sum(cell < 128) | |
| if dark_pixels < 10: | |
| rc.append(""); continue | |
| try: | |
| # Upscale cell (3x) | |
| ch, cw = cell.shape | |
| upscale_factor = 3 | |
| upscaled = cv2.resize(cell, (cw * upscale_factor, ch * upscale_factor), | |
| interpolation=cv2.INTER_CUBIC) | |
| # Apply CLAHE | |
| clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(4, 4)) | |
| enhanced = clahe.apply(upscaled) | |
| # Add padding | |
| pad = 15 | |
| eh, ew = enhanced.shape | |
| padded = np.ones((eh + pad * 2, ew + pad * 2), dtype=np.uint8) * 255 | |
| padded[pad:pad + eh, pad:pad + ew] = enhanced | |
| text = ocr_engine.ocr_region_best(padded, self.ocr_lang) | |
| rc.append(fix_persian_text(text).strip()) | |
| except Exception as e: | |
| rc.append("") | |
| cells.append(rc) | |
| non_empty_rows = [row for row in cells if any(c.strip() for c in row)] | |
| if not non_empty_rows: | |
| non_empty_rows = cells | |
| actual_rows = len(non_empty_rows) | |
| print(f" [TABLE] Extracted {actual_rows}x{cols} table (from {rows}x{cols} grid)") | |
| return TableData(bbox=table.bbox, rows=actual_rows, cols=cols, cells=non_empty_rows) | |
| def _find_positions(self, binary, axis, length): | |
| """Find line positions using morphological operations at multiple scales.""" | |
| positions = set() | |
| for scale in [3, 4, 5, 6]: | |
| try: | |
| k = max(length // scale, 10) | |
| if axis == 1: | |
| kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (k, 1)) | |
| mask = cv2.morphologyEx(binary, cv2.MORPH_OPEN, kernel) | |
| proj = np.sum(mask, axis=1) | |
| else: | |
| kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (1, k)) | |
| mask = cv2.morphologyEx(binary, cv2.MORPH_OPEN, kernel) | |
| proj = np.sum(mask, axis=0) | |
| thresh = np.max(proj) * 0.3 | |
| in_line = proj > thresh | |
| start = None | |
| for i, val in enumerate(in_line): | |
| if val: | |
| if start is None: start = i | |
| else: | |
| if start is not None: | |
| positions.add((start + i) // 2) | |
| start = None | |
| if start is not None: | |
| positions.add((start + len(in_line) - 1) // 2) | |
| except Exception: | |
| continue | |
| if not positions: | |
| return [] | |
| sp = sorted(positions) | |
| clusters = [[sp[0]]] | |
| for p in sp[1:]: | |
| if p - clusters[-1][-1] < 5: | |
| clusters[-1].append(p) | |
| else: | |
| clusters.append([p]) | |
| result = [int(np.mean(cl)) for cl in clusters] | |
| shape_len = binary.shape[0] if axis == 1 else binary.shape[1] | |
| if len(result) >= 2: | |
| avg_col_width = (result[-1] - result[0]) / (len(result) - 1) | |
| if result[0] > avg_col_width * 0.8: | |
| result.insert(0, 0) | |
| if shape_len - result[-1] > avg_col_width * 0.8: | |
| result.append(shape_len - 1) | |
| else: | |
| if not result or result[0] > shape_len * 0.15: | |
| result.insert(0, 0) | |
| if not result or result[-1] < shape_len * 0.85: | |
| result.append(shape_len - 1) | |
| return result | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # OCREngine β v6: Multi-PSM with confidence-based selection + better error handling | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class OCREngine: | |
| def __init__(self, lang="fas+ara+eng", dpi=300): | |
| self.lang = lang | |
| self.dpi = dpi | |
| def ocr_region(self, img, lang=None): | |
| """Simple OCR with PSM 6.""" | |
| lang = lang or self.lang | |
| try: | |
| if img.size == 0 or img.shape[0] < 3 or img.shape[1] < 3: | |
| return "" | |
| return pytesseract.image_to_string(Image.fromarray(img), lang=lang, | |
| config="--psm 6").strip() | |
| except Exception as e: | |
| print(f" [OCR] Warning: {e}") | |
| return "" | |
| def ocr_region_best(self, img, lang=None): | |
| """Multi-PSM OCR with confidence-based selection β picks best result.""" | |
| lang = lang or self.lang | |
| if img.size == 0 or img.shape[0] < 3 or img.shape[1] < 3: | |
| return "" | |
| pil = Image.fromarray(img) | |
| h, w = img.shape[:2] | |
| if h < 50 and w > h * 3: | |
| psms = [7, 13, 8] | |
| elif h < 80: | |
| psms = [7, 8, 6] | |
| else: | |
| psms = [3, 4, 6, 11] | |
| best_text, best_conf = "", -1 | |
| for psm in psms: | |
| try: | |
| data = pytesseract.image_to_data(pil, lang=lang, | |
| config=f'--psm {psm} --oem 3', | |
| output_type=pytesseract.Output.DICT) | |
| confs = [int(c) for c in data['conf'] if int(c) > 0] | |
| avg_conf = sum(confs) / len(confs) if confs else 0 | |
| text = pytesseract.image_to_string(pil, lang=lang, | |
| config=f'--psm {psm} --oem 3').strip() | |
| if avg_conf > best_conf: | |
| best_conf = avg_conf | |
| best_text = text | |
| except Exception: | |
| continue | |
| return best_text | |
| def ocr_full_page(self, img): | |
| """Full page OCR with PSM 3.""" | |
| try: | |
| return pytesseract.image_to_string(Image.fromarray(img), lang=self.lang, | |
| config="--psm 3").strip() | |
| except Exception as e: | |
| print(f" [OCR] Warning: {e}") | |
| return "" | |
| def ocr_with_hocr(self, img): | |
| """Generate hOCR via tesseract subprocess with 2x upscaling.""" | |
| try: | |
| if len(img.shape) == 2: | |
| work_img = cv2.resize(img, (img.shape[1]*2, img.shape[0]*2), | |
| interpolation=cv2.INTER_CUBIC) | |
| else: | |
| work_img = cv2.resize(img, (img.shape[1]*2, img.shape[0]*2), | |
| interpolation=cv2.INTER_CUBIC) | |
| img_path = _register_temp(".png") | |
| cv2.imwrite(img_path, work_img) | |
| effective_dpi = self.dpi * 2 | |
| result = subprocess.run( | |
| ['tesseract', img_path, 'stdout', '--dpi', str(effective_dpi), | |
| '--psm', '3', '-l', self.lang, 'hocr'], | |
| capture_output=True, text=True, timeout=60 | |
| ) | |
| # Cleanup immediately (atexit will catch stragglers) | |
| try: | |
| os.unlink(img_path) | |
| if img_path in _TEMP_FILES: | |
| _TEMP_FILES.remove(img_path) | |
| except Exception: | |
| pass | |
| return result.stdout | |
| except subprocess.TimeoutExpired: | |
| print(" [OCR] Warning: hOCR generation timed out") | |
| return "" | |
| except Exception as e: | |
| print(f" [OCR] Warning: hOCR generation failed: {e}") | |
| return "" | |
| def parse_hocr_blocks(self, hocr_text): | |
| """Parse hOCR into structured blocks with font size, bold, italic, confidence.""" | |
| blocks = [] | |
| if not hocr_text: | |
| return blocks | |
| try: | |
| soup = BeautifulSoup(hocr_text, "html.parser") | |
| except Exception: | |
| try: | |
| soup = BeautifulSoup(hocr_text, "xml") | |
| except Exception: | |
| return blocks | |
| paras = soup.find_all("p", class_="ocr_par") | |
| if not paras: | |
| paras = soup.find_all("span", class_="ocr_par") | |
| for par in paras: | |
| info = {"bbox": None, "text": "", "font_size": 12.0, | |
| "bold": False, "italic": False, "is_rtl": False, | |
| "confidence": 85.0, "alignment": "left", "words": []} | |
| info["bbox"] = self._parse_bbox(par) | |
| words = par.find_all("span", class_="ocrx_word") | |
| texts, sizes, confs, bolds = [], [], [], [] | |
| for w in words: | |
| t = w.get_text(strip=True) | |
| if not t: | |
| continue | |
| t = fix_persian_text(t) | |
| texts.append(t) | |
| fs = self._parse_font_size(w) | |
| conf = self._parse_confidence(w) | |
| info["words"].append({ | |
| "text": t, "bbox": self._parse_bbox(w), | |
| "font_size": fs, "bold": self._is_bold(w), | |
| "italic": self._is_italic(w), "confidence": conf}) | |
| if fs > 0: sizes.append(fs) | |
| if conf > 0: confs.append(conf) | |
| bolds.append(self._is_bold(w)) | |
| info["text"] = " ".join(texts) | |
| if sizes: | |
| info["font_size"] = max(set(sizes), key=sizes.count) | |
| if confs: | |
| info["confidence"] = sum(confs) / len(confs) | |
| info["bold"] = any(bolds) | |
| info["is_rtl"] = detect_rtl(info["text"]) | |
| if info["bbox"] and info["words"]: | |
| x_positions = [w.get("bbox", {}).get("x", 0) for w in info["words"] if w.get("bbox")] | |
| if x_positions: | |
| avg_x = sum(x_positions) / len(x_positions) | |
| bx = info["bbox"].get("x", 0) | |
| bw = info["bbox"].get("w", 1) | |
| if abs(avg_x - bx) < bw * 0.15: | |
| info["alignment"] = "center" | |
| if info["text"]: | |
| blocks.append(info) | |
| return blocks | |
| def _parse_bbox(self, el): | |
| title = el.get("title", "") | |
| if not title: return None | |
| m = re.search(r'bbox\s+(\d+)\s+(\d+)\s+(\d+)\s+(\d+)', title) | |
| if m: | |
| return {"x": int(m.group(1)), "y": int(m.group(2)), | |
| "w": int(m.group(3)) - int(m.group(1)), | |
| "h": int(m.group(4)) - int(m.group(2))} | |
| return None | |
| def _parse_font_size(self, el): | |
| title = el.get("title", "") | |
| m = re.search(r'x_fsize\s+([\d.]+)', title) | |
| if m: return float(m.group(1)) | |
| m = re.search(r'x_size\s+([\d.]+)', title) | |
| if m: return float(m.group(1)) | |
| bb = self._parse_bbox(el) | |
| if bb and bb["h"] > 0: return bb["h"] / 1.2 | |
| return 12.0 | |
| def _parse_confidence(self, el): | |
| title = el.get("title", "") | |
| m = re.search(r'x_wconf\s+(\d+)', title) | |
| if m: return float(m.group(1)) | |
| return 85.0 | |
| def _is_bold(self, el): | |
| title = str(el.get("title", "")).lower() | |
| return "bold" in title | |
| def _is_italic(self, el): | |
| title = str(el.get("title", "")).lower() | |
| return "italic" in title | |
| def get_overall_confidence(self, hocr_text): | |
| confs = [] | |
| try: | |
| soup = BeautifulSoup(hocr_text, "html.parser") | |
| except Exception: | |
| return 0.0 | |
| for w in soup.find_all("span", class_="ocrx_word"): | |
| m = re.search(r'x_wconf\s+(\d+)', w.get("title", "")) | |
| if m: confs.append(int(m.group(1))) | |
| return sum(confs) / len(confs) if confs else 0.0 | |
| def count_low_confidence_words(self, hocr_text, threshold=70): | |
| count = 0 | |
| try: | |
| soup = BeautifulSoup(hocr_text, "html.parser") | |
| except Exception: | |
| return 0 | |
| for w in soup.find_all("span", class_="ocrx_word"): | |
| m = re.search(r'x_wconf\s+(\d+)', w.get("title", "")) | |
| if m and int(m.group(1)) < threshold: | |
| count += 1 | |
| return count | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # ParagraphReconstructor β v6: Enhanced with error handling | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class ParagraphReconstructor: | |
| def __init__(self, median_font_size=12.0): | |
| self.median_font_size = median_font_size | |
| def reconstruct(self, blocks, column_boundaries=None): | |
| if not blocks: | |
| return [] | |
| blocks = [b for b in blocks if b.get("text", "").strip()] | |
| if not blocks: | |
| return [] | |
| is_rtl = detect_rtl(" ".join(b.get("text", "") for b in blocks)) | |
| if is_rtl: | |
| blocks.sort(key=lambda b: (b.get("bbox", {}).get("y", 0), | |
| -(b.get("bbox", {}).get("x", 0)))) | |
| else: | |
| blocks.sort(key=lambda b: (b.get("bbox", {}).get("y", 0), | |
| b.get("bbox", {}).get("x", 0))) | |
| paragraphs = [] | |
| current = blocks[0] | |
| for block in blocks[1:]: | |
| if self._should_merge(current, block): | |
| current = self._merge(current, block) | |
| else: | |
| paragraphs.append(current) | |
| current = block | |
| paragraphs.append(current) | |
| result = [] | |
| for para in paragraphs: | |
| bd = para.get("bbox") | |
| if not bd: continue | |
| try: | |
| tb = TextBlock( | |
| bbox=BBox(**bd), text=para.get("text", "").strip(), | |
| font_size=para.get("font_size", 12.0), | |
| is_bold=para.get("bold", False), | |
| is_italic=para.get("italic", False), | |
| is_rtl=para.get("is_rtl", False), | |
| confidence=para.get("confidence", 85.0), | |
| alignment=para.get("alignment", "left")) | |
| result.append(tb) | |
| except Exception as e: | |
| print(f" [RECON] Warning: Failed to create TextBlock: {e}") | |
| return result | |
| def _should_merge(self, b1, b2): | |
| bb1, bb2 = b1.get("bbox"), b2.get("bbox") | |
| if not bb1 or not bb2: return False | |
| gap = bb2["y"] - (bb1["y"] + bb1["h"]) | |
| max_gap = max(b1.get("font_size", 12) * 1.5, 15) | |
| if gap > max_gap or gap < -bb1["h"] * 0.5: return False | |
| xd = abs(bb1["x"] - bb2["x"]) | |
| if xd > max(bb1["w"], bb2["w"], 1) * 0.3: return False | |
| fs1, fs2 = b1.get("font_size", 12), b2.get("font_size", 12) | |
| if fs1 > 0 and fs2 > 0 and min(fs1, fs2) / max(fs1, fs2) < 0.7: return False | |
| if bb2["x"] > bb1["x"] + bb1["w"] * 1.2: return False | |
| return True | |
| def _merge(self, b1, b2): | |
| bb1, bb2 = b1.get("bbox", {}), b2.get("bbox", {}) | |
| x1 = min(bb1.get("x", 0), bb2.get("x", 0)) | |
| y1 = min(bb1.get("y", 0), bb2.get("y", 0)) | |
| x2 = max(bb1.get("x", 0) + bb1.get("w", 0), bb2.get("x", 0) + bb2.get("w", 0)) | |
| y2 = max(bb1.get("y", 0) + bb1.get("h", 0), bb2.get("y", 0) + bb2.get("h", 0)) | |
| return { | |
| "bbox": {"x": x1, "y": y1, "w": x2 - x1, "h": y2 - y1}, | |
| "text": b1.get("text", "") + " " + b2.get("text", ""), | |
| "font_size": max(b1.get("font_size", 12), b2.get("font_size", 12)), | |
| "bold": b1.get("bold", False) or b2.get("bold", False), | |
| "italic": b1.get("italic", False) or b2.get("italic", False), | |
| "is_rtl": b1.get("is_rtl", False) or b2.get("is_rtl", False), | |
| "confidence": min(b1.get("confidence", 85), b2.get("confidence", 85)), | |
| "words": b1.get("words", []) + b2.get("words", []), | |
| "alignment": b1.get("alignment", "left")} | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # HeadingDetector β v6: Enhanced multi-factor classification | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class HeadingDetector: | |
| def __init__(self, median_font_size=12.0): | |
| self.median_font_size = median_font_size | |
| def detect_font_sizes(self, paragraphs): | |
| sizes = [p.font_size for p in paragraphs if p.font_size > 0] | |
| if not sizes: return 12.0 | |
| ss = sorted(sizes) | |
| return ss[len(ss) // 2] | |
| def classify_paragraphs(self, paragraphs, page_width=2000): | |
| if not paragraphs: return paragraphs | |
| self.median_font_size = self.detect_font_sizes(paragraphs) | |
| bold_count = sum(1 for p in paragraphs if p.is_bold) | |
| center_count = sum(1 for p in paragraphs if p.alignment == "center") | |
| for p in paragraphs: | |
| p.element_type = self._classify(p, page_width, bold_count, center_count) | |
| return paragraphs | |
| def _classify(self, para, page_width, bold_count, center_count): | |
| fs = para.font_size | |
| text = para.text.strip() | |
| if not text: return "paragraph" | |
| if fs > self.median_font_size * 1.8: | |
| if para.alignment == "center" or len(text) < 100: | |
| return "title" | |
| if fs > self.median_font_size * 2.0: | |
| return "title" | |
| if fs > self.median_font_size * 1.4: | |
| return "heading" | |
| if fs > self.median_font_size * 1.2 and len(text) < 100: | |
| return "heading" | |
| if para.is_bold and len(text) < 80 and fs > self.median_font_size * 1.05: | |
| return "heading" | |
| if para.alignment == "center" and len(text) < 100 and fs > self.median_font_size * 1.05: | |
| return "subtitle" | |
| if para.is_bold and para.alignment == "center" and len(text) < 50: | |
| return "subtitle" | |
| if para.is_bold and para.bbox.y < page_width * 0.05 and len(text) < 60: | |
| return "heading" | |
| return "paragraph" | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # DocumentBuilder β v6: Professional Word document with enhanced styles | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class DocumentBuilder: | |
| def __init__(self, title=None, font_name="B Nazanin"): | |
| self.title = title | |
| self.font_name = font_name | |
| self.doc = None | |
| self._setup() | |
| def _setup(self): | |
| self.doc = Document() | |
| sec = self.doc.sections[0] | |
| sec.page_width = Cm(21.0); sec.page_height = Cm(29.7) | |
| sec.top_margin = Cm(2.5); sec.bottom_margin = Cm(2.5) | |
| sec.left_margin = Cm(2.5); sec.right_margin = Cm(2.5) | |
| style = self.doc.styles['Normal'] | |
| style.font.name = self.font_name; style.font.size = Pt(12) | |
| style.paragraph_format.space_after = Pt(6) | |
| style.paragraph_format.line_spacing = Pt(18) | |
| for lvl, sz in [('Heading 1', 18), ('Heading 2', 15), ('Heading 3', 13)]: | |
| try: | |
| s = self.doc.styles[lvl] | |
| s.font.name = self.font_name; s.font.size = Pt(sz); s.font.bold = True | |
| s.paragraph_format.space_before = Pt(12); s.paragraph_format.space_after = Pt(6) | |
| except Exception: | |
| pass | |
| footer = sec.footer | |
| footer.is_linked_to_previous = False | |
| p = footer.paragraphs[0] if footer.paragraphs else footer.add_paragraph() | |
| p.alignment = WD_ALIGN_PARAGRAPH.CENTER | |
| r = p.add_run() | |
| r._r.append(parse_xml(f'<w:fldChar {nsdecls("w")} w:fldCharType="begin"/>')) | |
| r2 = p.add_run() | |
| r2._r.append(parse_xml(f'<w:instrText {nsdecls("w")} xml:space="preserve"> PAGE </w:instrText>')) | |
| r3 = p.add_run() | |
| r3._r.append(parse_xml(f'<w:fldChar {nsdecls("w")} w:fldCharType="end"/>')) | |
| if self.title: | |
| h = self.doc.add_heading(self.title, level=0) | |
| h.alignment = WD_ALIGN_PARAGRAPH.CENTER | |
| for r in h.runs: | |
| r.font.name = self.font_name; r.font.size = Pt(22) | |
| def _rtl(self, p): | |
| p._p.get_or_add_pPr().set(qn('w:rtl'), '1') | |
| def _ltr(self, p): | |
| p._p.get_or_add_pPr().set(qn('w:ltr'), '1') | |
| def add_text_block(self, block): | |
| text = block.text.strip() | |
| if not text: return | |
| if block.element_type == "title": | |
| p = self.doc.add_heading(level=0); p.alignment = WD_ALIGN_PARAGRAPH.CENTER | |
| elif block.element_type == "heading": | |
| p = self.doc.add_heading(level=1) | |
| elif block.element_type == "subtitle": | |
| p = self.doc.add_heading(level=2); p.alignment = WD_ALIGN_PARAGRAPH.CENTER | |
| else: | |
| p = self.doc.add_paragraph() | |
| for r in p.runs: r.clear() | |
| r = p.add_run(text) | |
| r.font.name = self.font_name | |
| r.font.size = Pt(max(min(block.font_size, 72), 8)) | |
| if block.is_bold: r.bold = True | |
| if block.is_italic: r.italic = True | |
| if block.is_rtl: | |
| self._rtl(p); p.alignment = WD_ALIGN_PARAGRAPH.RIGHT | |
| else: | |
| self._ltr(p) | |
| if block.alignment == "center": p.alignment = WD_ALIGN_PARAGRAPH.CENTER | |
| pf = p.paragraph_format | |
| if block.element_type in ("title", "heading"): | |
| pf.space_before = Pt(12); pf.space_after = Pt(6) | |
| else: | |
| pf.space_before = Pt(3); pf.space_after = Pt(3) | |
| pf.line_spacing = Pt(block.font_size * 1.5) | |
| def add_table(self, td): | |
| if td.rows == 0 or td.cols == 0: return | |
| has = any(c.strip() for row in td.cells for c in row) | |
| if not has: return | |
| table = self.doc.add_table(rows=td.rows, cols=td.cols) | |
| table.style = 'Table Grid' | |
| table.alignment = WD_TABLE_ALIGNMENT.CENTER | |
| for r in range(td.rows): | |
| for c in range(td.cols): | |
| cell = table.cell(r, c); cell.text = "" | |
| tc = cell._tc | |
| tcPr = tc.get_or_add_tcPr() | |
| tcMar = parse_xml( | |
| f'<w:tcMar {nsdecls("w")}>' | |
| f' <w:top w:w="50" w:type="dxa"/>' | |
| f' <w:left w:w="100" w:type="dxa"/>' | |
| f' <w:bottom w:w="50" w:type="dxa"/>' | |
| f' <w:right w:w="100" w:type="dxa"/>' | |
| f'</w:tcMar>') | |
| tcPr.append(tcMar) | |
| if r < len(td.cells) and c < len(td.cells[r]): | |
| t = td.cells[r][c] | |
| if t: | |
| pp = cell.paragraphs[0] | |
| rr = pp.add_run(t) | |
| rr.font.name = self.font_name; rr.font.size = Pt(10) | |
| if detect_rtl(t): | |
| self._rtl(pp); pp.alignment = WD_ALIGN_PARAGRAPH.RIGHT | |
| else: | |
| self._ltr(pp) | |
| p = self.doc.add_paragraph() | |
| p.paragraph_format.space_before = Pt(3); p.paragraph_format.space_after = Pt(3) | |
| def add_line(self, line): | |
| if line.orientation == "horizontal": | |
| p = self.doc.add_paragraph() | |
| thick = min(max(line.thickness * 2, 4), 12) | |
| p._p.get_or_add_pPr().append(parse_xml( | |
| f'<w:pBdr {nsdecls("w")}><w:bottom w:val="single" w:sz="{thick}" w:space="1" w:color="000000"/></w:pBdr>')) | |
| p.paragraph_format.space_before = Pt(2); p.paragraph_format.space_after = Pt(2) | |
| def add_image_region(self, image_path, bbox, page_width=2000): | |
| if image_path and os.path.exists(image_path): | |
| try: | |
| w_in = min(bbox.w / 96.0, 6.0) | |
| self.doc.add_picture(image_path, width=Inches(w_in)) | |
| self.doc.paragraphs[-1].alignment = WD_ALIGN_PARAGRAPH.CENTER | |
| except Exception as e: | |
| print(f" [DOC] Warning: image insert failed: {e}") | |
| else: | |
| p = self.doc.add_paragraph("[ΨͺΨ΅ΩΫΨ±]") | |
| p.alignment = WD_ALIGN_PARAGRAPH.CENTER | |
| def add_page_break(self): | |
| self.doc.add_page_break() | |
| def save(self, output_path): | |
| self.doc.save(output_path) | |
| mb = os.path.getsize(output_path) / (1024 * 1024) | |
| print(f" [DOC] Saved: {output_path} ({mb:.2f} MB)") | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # QualityScorer β v6: Enhanced confidence-based quality scoring | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class QualityScorer: | |
| """Calculate quality score from 0 to 10 based on page statistics.""" | |
| def calculate(stats): | |
| """ | |
| Score formula (v6 enhanced): | |
| Base: 5.0 | |
| +1.0 if tables found | |
| +1.5 if confidence > 90, +1.0 if > 80, +0.5 if > 70 | |
| +0.5 if blocks > 2, +0.3 if blocks > 0 | |
| +1.0 if low_confidence_words == 0, +0.5 if < 3 | |
| +0.5 if lines found | |
| +0.5 if columns > 1 | |
| +0.5 if word count > 0 (text was found) | |
| Max: 10.0 | |
| """ | |
| score = 5.0 | |
| if stats.get('tables_found', 0) > 0: | |
| score += 1.0 | |
| conf = stats.get('confidence', 0) | |
| if conf > 90: | |
| score += 1.5 | |
| elif conf > 80: | |
| score += 1.0 | |
| elif conf > 70: | |
| score += 0.5 | |
| blocks = stats.get('blocks', 0) | |
| if blocks > 2: | |
| score += 0.5 | |
| elif blocks > 0: | |
| score += 0.3 | |
| low_conf = stats.get('low_conf_words', 0) | |
| if low_conf == 0: | |
| score += 1.0 | |
| elif low_conf < 3: | |
| score += 0.5 | |
| if stats.get('lines_found', 0) > 0: | |
| score += 0.5 | |
| if stats.get('columns', 1) > 1: | |
| score += 0.5 | |
| if stats.get('word_count', 0) > 0: | |
| score += 0.5 | |
| return min(score, 10.0) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # BatchProcessor β v6: PDF support, parallel processing, enhanced error handling | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class BatchProcessor: | |
| SUPPORTED = {'.jpg', '.jpeg', '.png', '.bmp', '.tiff', '.tif', '.webp', '.gif', '.pdf'} | |
| def __init__(self, lang="fas+ara+eng", dpi=300, title=None, | |
| font_name="B Nazanin", progress_callback=None, | |
| max_workers=None): | |
| self.lang = lang | |
| self.dpi = dpi | |
| self.title = title | |
| self.font_name = font_name | |
| self.progress_callback = progress_callback | |
| self.max_workers = max_workers or min(4, os.cpu_count() or 1) | |
| self.img_proc = ImageProcessor(dpi=dpi) | |
| self.ocr_engine = OCREngine(lang=lang, dpi=dpi) | |
| self.layout_analyzer = LayoutAnalyzer() | |
| self.table_extractor = TableExtractor(ocr_lang=lang) | |
| self.paragraph_reconstructor = ParagraphReconstructor() | |
| self.heading_detector = HeadingDetector() | |
| self.quality_scorer = QualityScorer() | |
| self.stats = [] | |
| def collect_images(self, inputs): | |
| """Collect images and PDFs from input paths. | |
| For PDFs, each page becomes a separate entry.""" | |
| images = [] | |
| for inp in inputs: | |
| p = Path(inp) | |
| if p.is_dir(): | |
| for ext in self.SUPPORTED: | |
| images.extend(str(x) for x in p.glob(f"*{ext}")) | |
| images.extend(str(x) for x in p.glob(f"*{ext.upper()}")) | |
| elif p.is_file() and p.suffix.lower() in self.SUPPORTED: | |
| if p.suffix.lower() == '.pdf': | |
| # For PDFs, expand to individual page entries | |
| pages = self._expand_pdf(str(p)) | |
| images.extend(pages) | |
| else: | |
| images.append(str(p)) | |
| else: | |
| print(f" [WARN] Skipping unsupported file: {inp}") | |
| # Sort: keep page-annotated entries grouped by original PDF | |
| images.sort(key=lambda x: (x.split('::')[0], int(x.split('::')[1]) if '::' in x else 0)) | |
| return list(dict.fromkeys(images)) | |
| def _expand_pdf(self, pdf_path): | |
| """Expand a PDF into per-page entries for batch processing. | |
| Returns list of 'pdf_path::page_index' strings.""" | |
| if not HAS_FITZ: | |
| print(f" [WARN] PDF support disabled (pymupdf not installed): {pdf_path}") | |
| return [] | |
| try: | |
| doc = fitz.open(pdf_path) | |
| count = len(doc) | |
| doc.close() | |
| return [f"{pdf_path}::{i}" for i in range(count)] | |
| except Exception as e: | |
| print(f" [WARN] Cannot read PDF {pdf_path}: {e}") | |
| return [] | |
| def _progress(self, cur, total, msg): | |
| if self.progress_callback: | |
| self.progress_callback(cur, total, msg) | |
| else: | |
| print(f" [{cur}/{total}] {msg}") | |
| def process_single_page(self, image_path, page_num, total): | |
| """Process a single page (image file or PDF page).""" | |
| ps = PageStatistics() | |
| try: | |
| print(f"\n{'=' * 60}") | |
| print(f" Page {page_num}/{total}: {os.path.basename(image_path)}") | |
| print(f"{'=' * 60}") | |
| t0 = time.time() | |
| # Handle PDF pages (encoded as 'pdf_path::page_index') | |
| if '::' in image_path and image_path.endswith(('::0', '::1', '::2', '::3', | |
| '::4', '::5', '::6', '::7', | |
| '::8', '::9')) is False: | |
| pass # Regular image | |
| elif '::' in image_path: | |
| pdf_path, page_idx = image_path.rsplit('::', 1) | |
| page_idx = int(page_idx) | |
| gray, binary = self._process_pdf_page(pdf_path, page_idx) | |
| else: | |
| gray, binary = self.img_proc.prepare_for_ocr(image_path) | |
| layout = self.layout_analyzer.analyze(binary, gray) | |
| hocr = self.ocr_engine.ocr_with_hocr(gray) | |
| hocr_blocks = self.ocr_engine.parse_hocr_blocks(hocr) | |
| print(f" [OCR] Found {len(hocr_blocks)} hOCR blocks") | |
| paragraphs = self.paragraph_reconstructor.reconstruct( | |
| hocr_blocks, layout.column_boundaries) | |
| paragraphs = self.heading_detector.classify_paragraphs( | |
| paragraphs, page_width=gray.shape[1]) | |
| layout.text_blocks = paragraphs | |
| # Extract tables with cell-by-cell OCR | |
| extracted = [] | |
| for t in layout.tables: | |
| try: | |
| extracted.append(self.table_extractor.extract_table(gray, t, self.ocr_engine)) | |
| except Exception as e: | |
| print(f" [TABLE] Warning: Table extraction failed: {e}") | |
| layout.tables = extracted | |
| oc = self.ocr_engine.get_overall_confidence(hocr) | |
| lwc = self.ocr_engine.count_low_confidence_words(hocr) | |
| self._build_page(layout, gray) | |
| ps.text_blocks = len(layout.text_blocks) | |
| ps.tables = len(layout.tables) | |
| ps.lines = len([l for l in layout.lines if not l.is_table_border]) | |
| ps.image_regions = len(layout.image_regions) | |
| ps.columns = layout.columns | |
| ps.overall_confidence = oc | |
| ps.is_rtl = layout.is_rtl | |
| ps.total_words = sum(len(b.text.split()) for b in layout.text_blocks) | |
| ps.low_confidence_words = lwc | |
| ps.processing_time = time.time() - t0 | |
| print(f" [STATS] {ps.text_blocks} blocks, {ps.tables} tables, " | |
| f"{ps.lines} lines, conf={oc:.0f}%, {ps.processing_time:.1f}s") | |
| if lwc: print(f" [WARN] {lwc} low-confidence words") | |
| page_stats = { | |
| 'tables_found': ps.tables, | |
| 'confidence': oc, | |
| 'blocks': ps.text_blocks, | |
| 'low_conf_words': lwc, | |
| 'lines_found': ps.lines, | |
| 'columns': ps.columns, | |
| 'word_count': ps.total_words, | |
| } | |
| page_quality = self.quality_scorer.calculate(page_stats) | |
| print(f" [QUALITY] Page score: {page_quality}/10") | |
| self._progress(page_num, total, f"{ps.text_blocks} blocks, {ps.tables} tables, quality={page_quality}/10") | |
| return ps | |
| except Exception as e: | |
| print(f" [ERROR] Page {page_num}: {e}") | |
| import traceback; traceback.print_exc() | |
| return ps | |
| def _process_pdf_page(self, pdf_path, page_idx): | |
| """Convert a single PDF page to grayscale and binary images.""" | |
| if not HAS_FITZ: | |
| raise ImportError("pymupdf required for PDF support") | |
| doc = fitz.open(pdf_path) | |
| page = doc[page_idx] | |
| zoom = self.dpi / 72.0 | |
| mat = fitz.Matrix(zoom, zoom) | |
| pix = page.get_pixmap(matrix=mat, alpha=False) | |
| img_data = np.frombuffer(pix.samples, dtype=np.uint8) | |
| img = img_data.reshape(pix.height, pix.width, 3) | |
| img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) | |
| doc.close() | |
| print(f" [PDF] Page {page_idx + 1}: {pix.width}x{pix.height}") | |
| return self.img_proc.prepare_for_ocr_from_array(img) | |
| def _build_page(self, layout, gray): | |
| elements = [] | |
| for b in layout.text_blocks: | |
| if b.text.strip(): | |
| elements.append(("text", b.bbox.y, b)) | |
| for t in layout.tables: | |
| elements.append(("table", t.bbox.y, t)) | |
| for l in layout.lines: | |
| if not l.is_table_border: | |
| elements.append(("line", l.bbox.y, l)) | |
| for ir in layout.image_regions: | |
| elements.append(("image", ir.bbox.y, ir)) | |
| elements.sort(key=lambda e: e[1]) | |
| for et, yp, elem in elements: | |
| try: | |
| if et == "text": self.doc_builder.add_text_block(elem) | |
| elif et == "table": self.doc_builder.add_table(elem) | |
| elif et == "line" and elem.orientation == "horizontal": | |
| self.doc_builder.add_line(elem) | |
| elif et == "image": | |
| ip = self._extract_img(gray, elem) | |
| self.doc_builder.add_image_region(ip, elem.bbox, gray.shape[1]) | |
| except Exception as e: | |
| print(f" [DOC] Warning: Failed to add element: {e}") | |
| def _extract_img(self, gray, ir): | |
| try: | |
| h, w = gray.shape | |
| x1, y1 = max(0, ir.bbox.x), max(0, ir.bbox.y) | |
| x2, y2 = min(w, ir.bbox.x2), min(h, ir.bbox.y2) | |
| if x2 <= x1 or y2 <= y1: return None | |
| region = gray[y1:y2, x1:x2] | |
| tmp_path = _register_temp(".png") | |
| cv2.imwrite(tmp_path, region) | |
| return tmp_path | |
| except Exception: | |
| return None | |
| def process_batch(self, input_paths, output_path, title=None): | |
| """Process multiple images/PDFs into a single Word document.""" | |
| images = self.collect_images(input_paths) | |
| if not images: | |
| print("[ERROR] No valid images or PDFs!") | |
| return False | |
| total = len(images) | |
| print(f"\n{'#' * 60}") | |
| print(f" Scan2Doc Pro v6.0 β OCR-to-Word with PDF Support") | |
| print(f" {total} page(s), langs={self.lang}, output={output_path}") | |
| if not HAS_FITZ: | |
| print(f" [WARN] pymupdf not installed β PDF support disabled") | |
| print(f"{'#' * 60}") | |
| self.doc_builder = DocumentBuilder(title=title, font_name=self.font_name) | |
| ok, fail = 0, [] | |
| t0 = time.time() | |
| for i, ip in enumerate(images, 1): | |
| try: | |
| ps = self.process_single_page(ip, i, total) | |
| if ps.text_blocks > 0 or ps.tables > 0: | |
| ok += 1; self.stats.append(ps) | |
| else: | |
| fail.append(i) | |
| except Exception as e: | |
| print(f" [ERROR] Page {i}: {e}") | |
| import traceback; traceback.print_exc() | |
| fail.append(i) | |
| if i < total: self.doc_builder.add_page_break() | |
| try: | |
| self.doc_builder.save(output_path) | |
| except Exception as e: | |
| print(f"[ERROR] Save failed: {e}") | |
| return False | |
| tt = time.time() - t0 | |
| print(f"\n{'#' * 60}") | |
| print(f" Done! {ok}/{total} pages, {tt:.1f}s") | |
| if fail: print(f" Failed: {fail}") | |
| if self.stats: | |
| ac = sum(s.overall_confidence for s in self.stats) / len(self.stats) | |
| tb = sum(s.text_blocks for s in self.stats) | |
| tt2 = sum(s.tables for s in self.stats) | |
| tl = sum(s.lines for s in self.stats) | |
| tw = sum(s.total_words for s in self.stats) | |
| lc = sum(s.low_confidence_words for s in self.stats) | |
| print(f"\n === AGGREGATE STATISTICS ===") | |
| print(f" Blocks: {tb}") | |
| print(f" Tables: {tt2}") | |
| print(f" Lines: {tl}") | |
| print(f" Words: {tw}") | |
| print(f" Confidence: {ac:.1f}%") | |
| print(f" Low-confidence words: {lc}") | |
| agg_stats = { | |
| 'tables_found': tt2, | |
| 'confidence': ac, | |
| 'blocks': tb, | |
| 'low_conf_words': lc, | |
| 'lines_found': tl, | |
| 'columns': max((s.columns for s in self.stats), default=1), | |
| 'word_count': tw, | |
| } | |
| quality = self.quality_scorer.calculate(agg_stats) | |
| print(f"\n === QUALITY SCORE: {quality}/10 ===") | |
| print(f"{'#' * 60}") | |
| return ok > 0 | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # CLI β v6: Same interface as v5, with PDF support | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def build_parser(): | |
| p = argparse.ArgumentParser( | |
| prog="scan2doc_pro_v6", | |
| description="Scan2Doc Pro v6.0 β OCR-to-Word Converter with PDF Support") | |
| p.add_argument("-i", "--input", nargs="+", required=True, | |
| help="Input image(s) or PDF(s)") | |
| p.add_argument("-o", "--output", required=True, help="Output .docx path") | |
| p.add_argument("--lang", default="fas+ara+eng", help="Tesseract language(s)") | |
| p.add_argument("--dpi", type=int, default=300, help="Scan DPI") | |
| p.add_argument("--title", default=None, help="Document title") | |
| p.add_argument("--font", default="B Nazanin", help="Font name") | |
| p.add_argument("--workers", type=int, default=None, | |
| help="Max parallel workers (default: auto)") | |
| return p | |
| def main(): | |
| args = build_parser().parse_args() | |
| if not args.output.lower().endswith(".docx"): | |
| args.output += ".docx" | |
| proc = BatchProcessor(lang=args.lang, dpi=args.dpi, title=args.title, | |
| font_name=args.font, max_workers=args.workers) | |
| ok = proc.process_batch(args.input, args.output, title=args.title) | |
| sys.exit(0 if ok else 1) | |
| if __name__ == "__main__": | |
| main() | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # GUI Section | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| #!/usr/bin/env python3 | |
| """ | |
| Scan2Doc Pro β Professional GUI Application | |
| ============================================ | |
| Modern CustomTkinter-based desktop application for OCR-to-Word conversion. | |
| Features: image preview, real-time logging, batch processing, quality scoring. | |
| """ | |
| import os | |
| import sys | |
| import threading | |
| import time | |
| import traceback | |
| from pathlib import Path | |
| # ββ Headless-safe import for customtkinter ββ | |
| os.environ.setdefault("DISPLAY", ":0") | |
| try: | |
| import customtkinter as ctk | |
| except ImportError: | |
| print("ERROR: customtkinter not installed. Run: pip install customtkinter") | |
| sys.exit(1) | |
| # Set theme | |
| ctk.set_appearance_mode("dark") | |
| ctk.set_default_color_theme("blue") | |
| try: | |
| from PIL import Image, ImageTk | |
| PIL_AVAILABLE = True | |
| except ImportError: | |
| PIL_AVAILABLE = False | |
| # Import OCR engine | |
| sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) | |
| try: | |
| from scan2doc_pro_v6 import BatchProcessor, QualityScorer, PageStatistics | |
| except ImportError: | |
| # Try alternate path | |
| try: | |
| sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) | |
| from scan2doc_pro_v6 import BatchProcessor, QualityScorer, PageStatistics | |
| except ImportError: | |
| print("ERROR: Cannot import scan2doc_pro_v5.py") | |
| print("Make sure scan2doc_pro_v5.py is in the same directory as this script.") | |
| sys.exit(1) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Color Palette | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| COLORS = { | |
| "bg_dark": "#1a1a2e", | |
| "bg_medium": "#16213e", | |
| "bg_light": "#0f3460", | |
| "accent": "#e94560", | |
| "accent2": "#533483", | |
| "text": "#eaeaea", | |
| "text_dim": "#8899aa", | |
| "success": "#00c853", | |
| "warning": "#ffab00", | |
| "error": "#ff1744", | |
| "sidebar": "#0f0f23", | |
| "card": "#1a1a3e", | |
| "blue": "#2979ff", | |
| "blue_dark": "#1565c0", | |
| } | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # File List Item Widget | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class FileListItem(ctk.CTkFrame): | |
| """Individual file item in the file list.""" | |
| def __init__(self, master, filepath, on_remove=None, **kwargs): | |
| super().__init__(master, fg_color=COLORS["card"], corner_radius=6, **kwargs) | |
| self.filepath = filepath | |
| self.on_remove = on_remove | |
| self.grid_columnconfigure(0, weight=1) | |
| # Filename label | |
| name = os.path.basename(filepath) | |
| if len(name) > 30: | |
| name = name[:27] + "..." | |
| self.label = ctk.CTkLabel( | |
| self, text=name, anchor="w", | |
| text_color=COLORS["text"], font=ctk.CTkFont(size=11)) | |
| self.label.grid(row=0, column=0, padx=(8, 4), pady=4, sticky="ew") | |
| # Remove button | |
| self.remove_btn = ctk.CTkButton( | |
| self, text="β", width=24, height=24, | |
| fg_color="transparent", hover_color=COLORS["error"], | |
| text_color=COLORS["text_dim"], font=ctk.CTkFont(size=12, weight="bold"), | |
| command=self._remove) | |
| self.remove_btn.grid(row=0, column=1, padx=(0, 6), pady=4) | |
| def _remove(self): | |
| if self.on_remove: | |
| self.on_remove(self.filepath) | |
| self.destroy() | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Sidebar Panel | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class Sidebar(ctk.CTkFrame): | |
| """Left sidebar with file management and settings.""" | |
| def __init__(self, master, app, **kwargs): | |
| super().__init__(master, fg_color=COLORS["sidebar"], corner_radius=0, width=280, **kwargs) | |
| self.app = app | |
| self.pack_propagate(False) | |
| self.grid_propagate(False) | |
| self.files = [] | |
| self._build_ui() | |
| def _build_ui(self): | |
| # ββ Logo / Title ββ | |
| title_frame = ctk.CTkFrame(self, fg_color="transparent") | |
| title_frame.pack(fill="x", padx=15, pady=(15, 5)) | |
| ctk.CTkLabel( | |
| title_frame, text="π Scan2Doc Pro", | |
| font=ctk.CTkFont(size=20, weight="bold"), | |
| text_color=COLORS["blue"] | |
| ).pack(anchor="w") | |
| ctk.CTkLabel( | |
| title_frame, text="Professional OCR β Word", | |
| font=ctk.CTkFont(size=11), | |
| text_color=COLORS["text_dim"] | |
| ).pack(anchor="w") | |
| # Separator | |
| ctk.CTkFrame(self, fg_color=COLORS["bg_light"], height=1).pack(fill="x", padx=15, pady=10) | |
| # ββ File Selection ββ | |
| ctk.CTkLabel(self, text="π Input Files", font=ctk.CTkFont(size=13, weight="bold"), | |
| text_color=COLORS["text"]).pack(anchor="w", padx=15, pady=(5, 5)) | |
| btn_frame = ctk.CTkFrame(self, fg_color="transparent") | |
| btn_frame.pack(fill="x", padx=15) | |
| ctk.CTkButton( | |
| btn_frame, text="π Select Folder", height=32, | |
| fg_color=COLORS["blue_dark"], hover_color=COLORS["blue"], | |
| font=ctk.CTkFont(size=12), command=self._select_folder | |
| ).pack(side="left", expand=True, fill="x", padx=(0, 4)) | |
| ctk.CTkButton( | |
| btn_frame, text="π Add Files", height=32, | |
| fg_color=COLORS["accent2"], hover_color=COLORS["accent"], | |
| font=ctk.CTkFont(size=12), command=self._select_files | |
| ).pack(side="left", expand=True, fill="x", padx=(4, 0)) | |
| # ββ File List ββ | |
| self.file_list_frame = ctk.CTkScrollableFrame( | |
| self, fg_color=COLORS["bg_medium"], corner_radius=8, | |
| height=120) | |
| self.file_list_frame.pack(fill="x", padx=15, pady=(8, 5)) | |
| self.file_count_label = ctk.CTkLabel( | |
| self, text="No files selected", font=ctk.CTkFont(size=10), | |
| text_color=COLORS["text_dim"]) | |
| self.file_count_label.pack(anchor="w", padx=15) | |
| # Clear all button | |
| ctk.CTkButton( | |
| self, text="π Clear All", height=24, width=80, | |
| fg_color="transparent", border_width=1, border_color=COLORS["error"], | |
| text_color=COLORS["error"], hover_color="#330000", | |
| font=ctk.CTkFont(size=10), command=self._clear_files | |
| ).pack(anchor="e", padx=15, pady=(2, 8)) | |
| # Separator | |
| ctk.CTkFrame(self, fg_color=COLORS["bg_light"], height=1).pack(fill="x", padx=15, pady=5) | |
| # ββ Settings ββ | |
| ctk.CTkLabel(self, text="βοΈ Settings", font=ctk.CTkFont(size=13, weight="bold"), | |
| text_color=COLORS["text"]).pack(anchor="w", padx=15, pady=(5, 8)) | |
| # DPI | |
| self._setting_label("DPI:") | |
| self.dpi_var = ctk.StringVar(value="300") | |
| ctk.CTkOptionMenu( | |
| self, variable=self.dpi_var, | |
| values=["150", "200", "300", "400", "500", "600"], | |
| fg_color=COLORS["bg_medium"], button_color=COLORS["blue_dark"], | |
| width=120, height=28, font=ctk.CTkFont(size=11) | |
| ).pack(anchor="w", padx=15, pady=(0, 6)) | |
| # Language | |
| self._setting_label("Language:") | |
| self.lang_var = ctk.StringVar(value="Persian + English") | |
| ctk.CTkOptionMenu( | |
| self, variable=self.lang_var, | |
| values=["Persian + English", "Persian", "English", "Arabic", "All Languages"], | |
| fg_color=COLORS["bg_medium"], button_color=COLORS["blue_dark"], | |
| width=160, height=28, font=ctk.CTkFont(size=11) | |
| ).pack(anchor="w", padx=15, pady=(0, 6)) | |
| # Title | |
| self._setting_label("Document Title:") | |
| self.title_entry = ctk.CTkEntry( | |
| self, placeholder_text="Optional title...", | |
| fg_color=COLORS["bg_medium"], border_color=COLORS["blue_dark"], | |
| height=28, font=ctk.CTkFont(size=11)) | |
| self.title_entry.pack(fill="x", padx=15, pady=(0, 6)) | |
| # Checkboxes | |
| self.tables_var = ctk.BooleanVar(value=True) | |
| self.lines_var = ctk.BooleanVar(value=True) | |
| self.images_var = ctk.BooleanVar(value=False) | |
| checks_frame = ctk.CTkFrame(self, fg_color="transparent") | |
| checks_frame.pack(fill="x", padx=15, pady=(0, 4)) | |
| ctk.CTkCheckBox( | |
| checks_frame, text="Tables", variable=self.tables_var, | |
| text_color=COLORS["text"], font=ctk.CTkFont(size=11), | |
| fg_color=COLORS["blue_dark"], hover_color=COLORS["blue"], | |
| checkmark_color=COLORS["text"] | |
| ).pack(side="left", padx=(0, 8)) | |
| ctk.CTkCheckBox( | |
| checks_frame, text="Lines", variable=self.lines_var, | |
| text_color=COLORS["text"], font=ctk.CTkFont(size=11), | |
| fg_color=COLORS["blue_dark"], hover_color=COLORS["blue"], | |
| checkmark_color=COLORS["text"] | |
| ).pack(side="left", padx=(0, 8)) | |
| ctk.CTkCheckBox( | |
| checks_frame, text="Images", variable=self.images_var, | |
| text_color=COLORS["text"], font=ctk.CTkFont(size=11), | |
| fg_color=COLORS["blue_dark"], hover_color=COLORS["blue"], | |
| checkmark_color=COLORS["text"] | |
| ).pack(side="left") | |
| # Separator | |
| ctk.CTkFrame(self, fg_color=COLORS["bg_light"], height=1).pack(fill="x", padx=15, pady=8) | |
| # ββ Output ββ | |
| ctk.CTkLabel(self, text="πΎ Output", font=ctk.CTkFont(size=13, weight="bold"), | |
| text_color=COLORS["text"]).pack(anchor="w", padx=15, pady=(0, 5)) | |
| self.output_var = ctk.StringVar(value="output.docx") | |
| self.output_entry = ctk.CTkEntry( | |
| self, textvariable=self.output_var, | |
| fg_color=COLORS["bg_medium"], border_color=COLORS["blue_dark"], | |
| height=28, font=ctk.CTkFont(size=11)) | |
| self.output_entry.pack(fill="x", padx=15, pady=(0, 4)) | |
| ctk.CTkButton( | |
| self, text="π Browse...", height=26, | |
| fg_color=COLORS["bg_medium"], border_width=1, | |
| border_color=COLORS["blue_dark"], hover_color=COLORS["blue_dark"], | |
| font=ctk.CTkFont(size=11), command=self._select_output | |
| ).pack(fill="x", padx=15, pady=(0, 10)) | |
| # ββ Start Button ββ | |
| self.start_btn = ctk.CTkButton( | |
| self, text="βΆ START PROCESSING", height=44, | |
| fg_color=COLORS["blue"], hover_color=COLORS["blue_dark"], | |
| font=ctk.CTkFont(size=15, weight="bold"), | |
| corner_radius=10, command=self._start_processing) | |
| self.start_btn.pack(fill="x", padx=15, pady=(0, 15)) | |
| def _setting_label(self, text): | |
| ctk.CTkLabel(self, text=text, font=ctk.CTkFont(size=11), | |
| text_color=COLORS["text_dim"]).pack(anchor="w", padx=15, pady=(0, 2)) | |
| def _select_folder(self): | |
| try: | |
| from tkinter import filedialog | |
| folder = filedialog.askdirectory(title="Select Image Folder") | |
| if folder: | |
| exts = {'.jpg', '.jpeg', '.png', '.bmp', '.tiff', '.tif', '.webp', '.gif'} | |
| files = [] | |
| for f in sorted(Path(folder).iterdir()): | |
| if f.suffix.lower() in exts: | |
| files.append(str(f)) | |
| if files: | |
| self.files.extend(files) | |
| self._refresh_file_list() | |
| self.app.log(f"Added {len(files)} files from folder") | |
| else: | |
| self.app.log("No image files found in selected folder", "warning") | |
| except Exception as e: | |
| self.app.log(f"Error selecting folder: {e}", "error") | |
| def _select_files(self): | |
| try: | |
| from tkinter import filedialog | |
| files = filedialog.askopenfilenames( | |
| title="Select Image Files", | |
| filetypes=[ | |
| ("Image files", "*.jpg *.jpeg *.png *.bmp *.tiff *.tif *.webp *.gif"), | |
| ("All files", "*.*") | |
| ]) | |
| if files: | |
| self.files.extend(list(files)) | |
| self._refresh_file_list() | |
| self.app.log(f"Added {len(files)} file(s)") | |
| except Exception as e: | |
| self.app.log(f"Error selecting files: {e}", "error") | |
| def _select_output(self): | |
| try: | |
| from tkinter import filedialog | |
| path = filedialog.asksaveasfilename( | |
| title="Save Output As", | |
| defaultextension=".docx", | |
| filetypes=[("Word Document", "*.docx")]) | |
| if path: | |
| self.output_var.set(path) | |
| except Exception as e: | |
| self.app.log(f"Error: {e}", "error") | |
| def _clear_files(self): | |
| self.files.clear() | |
| self._refresh_file_list() | |
| self.app.log("File list cleared") | |
| def _remove_file(self, filepath): | |
| if filepath in self.files: | |
| self.files.remove(filepath) | |
| self._refresh_file_list() | |
| def _refresh_file_list(self): | |
| # Clear existing items | |
| for widget in self.file_list_frame.winfo_children(): | |
| widget.destroy() | |
| for fp in self.files: | |
| item = FileListItem(self.file_list_frame, fp, on_remove=self._remove_file) | |
| item.pack(fill="x", padx=2, pady=2) | |
| count = len(self.files) | |
| self.file_count_label.configure( | |
| text=f"{count} file(s) selected" if count > 0 else "No files selected") | |
| # Auto-set output filename | |
| if count == 1: | |
| stem = Path(self.files[0]).stem | |
| self.output_var.set(f"{stem}.docx") | |
| elif count > 1: | |
| parent = os.path.basename(os.path.dirname(self.files[0])) | |
| if parent: | |
| self.output_var.set(f"{parent}.docx") | |
| def get_settings(self): | |
| """Return current settings as a dict.""" | |
| lang_map = { | |
| "Persian + English": "fas+eng", | |
| "Persian": "fas", | |
| "English": "eng", | |
| "Arabic": "ara", | |
| "All Languages": "fas+ara+eng" | |
| } | |
| return { | |
| "files": self.files[:], | |
| "dpi": int(self.dpi_var.get()), | |
| "lang": lang_map.get(self.lang_var.get(), "fas+eng"), | |
| "title": self.title_entry.get().strip() or None, | |
| "output": self.output_var.get().strip() or "output.docx", | |
| "tables": self.tables_var.get(), | |
| "lines": self.lines_var.get(), | |
| "images": self.images_var.get(), | |
| } | |
| def set_processing_state(self, processing): | |
| """Enable/disable UI during processing.""" | |
| state = "disabled" if processing else "normal" | |
| self.start_btn.configure( | |
| state=state, | |
| text="β³ Processing..." if processing else "βΆ START PROCESSING") | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Main Content Area | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class MainArea(ctk.CTkFrame): | |
| """Right main area with preview, log, and progress.""" | |
| def __init__(self, master, app, **kwargs): | |
| super().__init__(master, fg_color=COLORS["bg_dark"], corner_radius=0, **kwargs) | |
| self.app = app | |
| self.grid_rowconfigure(1, weight=1) | |
| self.grid_columnconfigure(0, weight=1) | |
| self._build_ui() | |
| def _build_ui(self): | |
| # ββ Preview Panel ββ | |
| preview_label = ctk.CTkLabel( | |
| self, text="πΌοΈ Preview", font=ctk.CTkFont(size=13, weight="bold"), | |
| text_color=COLORS["text"]) | |
| preview_label.grid(row=0, column=0, padx=15, pady=(10, 5), sticky="w") | |
| self.preview_frame = ctk.CTkFrame( | |
| self, fg_color=COLORS["bg_medium"], corner_radius=10, | |
| height=250) | |
| self.preview_frame.grid(row=0, column=0, padx=15, pady=(30, 5), sticky="nsew") | |
| self.preview_frame.grid_propagate(False) | |
| self.preview_label = ctk.CTkLabel( | |
| self.preview_frame, text="Select files to preview", | |
| text_color=COLORS["text_dim"], font=ctk.CTkFont(size=12)) | |
| self.preview_label.pack(expand=True) | |
| # ββ Processing Log ββ | |
| log_label = ctk.CTkLabel( | |
| self, text="π Processing Log", font=ctk.CTkFont(size=13, weight="bold"), | |
| text_color=COLORS["text"]) | |
| log_label.grid(row=1, column=0, padx=15, pady=(10, 5), sticky="nw") | |
| self.log_text = ctk.CTkTextbox( | |
| self, fg_color=COLORS["bg_medium"], text_color=COLORS["text"], | |
| font=ctk.CTkFont(family="Consolas", size=11), | |
| corner_radius=10, wrap="word") | |
| self.log_text.grid(row=1, column=0, padx=15, pady=(30, 5), sticky="nsew") | |
| self.log_text.configure(state="disabled") | |
| # ββ Footer: Progress + Status ββ | |
| footer_frame = ctk.CTkFrame(self, fg_color=COLORS["bg_medium"], corner_radius=10) | |
| footer_frame.grid(row=2, column=0, padx=15, pady=(5, 10), sticky="ew") | |
| footer_frame.grid_columnconfigure(0, weight=1) | |
| # Progress bar | |
| self.progress_bar = ctk.CTkProgressBar( | |
| footer_frame, progress_color=COLORS["blue"], | |
| fg_color=COLORS["bg_dark"], height=12) | |
| self.progress_bar.grid(row=0, column=0, padx=10, pady=(8, 4), sticky="ew") | |
| self.progress_bar.set(0) | |
| # Status labels | |
| status_frame = ctk.CTkFrame(footer_frame, fg_color="transparent") | |
| status_frame.grid(row=1, column=0, padx=10, pady=(0, 8), sticky="ew") | |
| status_frame.grid_columnconfigure(0, weight=1) | |
| status_frame.grid_columnconfigure(1, weight=1) | |
| status_frame.grid_columnconfigure(2, weight=1) | |
| self.progress_label = ctk.CTkLabel( | |
| status_frame, text="0%", font=ctk.CTkFont(size=12, weight="bold"), | |
| text_color=COLORS["blue"]) | |
| self.progress_label.grid(row=0, column=0, sticky="w") | |
| self.page_label = ctk.CTkLabel( | |
| status_frame, text="0 / 0 pages", font=ctk.CTkFont(size=11), | |
| text_color=COLORS["text_dim"]) | |
| self.page_label.grid(row=0, column=1, sticky="w") | |
| self.quality_label = ctk.CTkLabel( | |
| status_frame, text="Quality: --", font=ctk.CTkFont(size=11, weight="bold"), | |
| text_color=COLORS["text_dim"]) | |
| self.quality_label.grid(row=0, column=2, sticky="e") | |
| def log(self, message, level="info"): | |
| """Thread-safe log message.""" | |
| prefix = {"info": "βΉοΈ", "warning": "β οΈ", "error": "β", "success": "β "}.get(level, "βΉοΈ") | |
| color_map = {"info": COLORS["text"], "warning": COLORS["warning"], | |
| "error": COLORS["error"], "success": COLORS["success"]} | |
| timestamp = time.strftime("%H:%M:%S") | |
| line = f"[{timestamp}] {prefix} {message}\n" | |
| def _append(): | |
| self.log_text.configure(state="normal") | |
| self.log_text.insert("end", line) | |
| self.log_text.see("end") | |
| self.log_text.configure(state="disabled") | |
| try: | |
| self.after(0, _append) | |
| except Exception: | |
| pass | |
| def update_preview(self, filepath): | |
| """Update the preview panel with an image.""" | |
| if not PIL_AVAILABLE: | |
| return | |
| def _update(): | |
| try: | |
| for widget in self.preview_frame.winfo_children(): | |
| widget.destroy() | |
| img = Image.open(filepath) | |
| # Calculate fit size | |
| frame_w = self.preview_frame.winfo_width() - 20 | |
| frame_h = self.preview_frame.winfo_height() - 20 | |
| if frame_w < 10: frame_w = 400 | |
| if frame_h < 10: frame_h = 200 | |
| img.thumbnail((frame_w, frame_h), Image.Resampling.LANCZOS) | |
| photo = ImageTk.PhotoImage(img) | |
| label = ctk.CTkLabel(self.preview_frame, image=photo, text="") | |
| label.image = photo # Keep reference | |
| label.pack(expand=True) | |
| except Exception as e: | |
| self.log(f"Preview error: {e}", "warning") | |
| try: | |
| self.after(0, _update) | |
| except Exception: | |
| pass | |
| def update_progress(self, current, total, message=""): | |
| """Update progress bar and labels.""" | |
| def _update(): | |
| pct = (current / total * 100) if total > 0 else 0 | |
| self.progress_bar.set(current / total if total > 0 else 0) | |
| self.progress_label.configure(text=f"{pct:.0f}%") | |
| self.page_label.configure(text=f"{current} / {total} pages") | |
| try: | |
| self.after(0, _update) | |
| except Exception: | |
| pass | |
| def update_quality(self, score): | |
| """Update quality score display.""" | |
| def _update(): | |
| if score >= 9: | |
| color = COLORS["success"] | |
| elif score >= 7: | |
| color = COLORS["blue"] | |
| elif score >= 5: | |
| color = COLORS["warning"] | |
| else: | |
| color = COLORS["error"] | |
| self.quality_label.configure(text=f"Quality: {score}/10", text_color=color) | |
| try: | |
| self.after(0, _update) | |
| except Exception: | |
| pass | |
| def reset(self): | |
| """Reset progress display.""" | |
| self.progress_bar.set(0) | |
| self.progress_label.configure(text="0%", text_color=COLORS["blue"]) | |
| self.page_label.configure(text="0 / 0 pages") | |
| self.quality_label.configure(text="Quality: --", text_color=COLORS["text_dim"]) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Main Application Window | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class Scan2DocApp(ctk.CTk): | |
| """Main application window.""" | |
| def __init__(self): | |
| super().__init__() | |
| self.title("Scan2Doc Pro β Professional OCR to Word") | |
| self.geometry("1280x800") | |
| self.minsize(900, 600) | |
| self.configure(fg_color=COLORS["bg_dark"]) | |
| self.processing = False | |
| self._build_layout() | |
| def _build_layout(self): | |
| # Grid layout: sidebar (fixed) + main area (expandable) | |
| self.grid_columnconfigure(1, weight=1) | |
| self.grid_rowconfigure(0, weight=1) | |
| # Sidebar | |
| self.sidebar = Sidebar(self, self) | |
| self.sidebar.grid(row=0, column=0, sticky="ns") | |
| # Main area | |
| self.main_area = MainArea(self, self) | |
| self.main_area.grid(row=0, column=1, sticky="nsew") | |
| def log(self, message, level="info"): | |
| """Log a message to the processing log.""" | |
| self.main_area.log(message, level) | |
| def _start_processing(self): | |
| """Start OCR processing in a background thread.""" | |
| settings = self.sidebar.get_settings() | |
| if not settings["files"]: | |
| self.log("No files selected! Please add images first.", "error") | |
| return | |
| if self.processing: | |
| return | |
| self.processing = True | |
| self.sidebar.set_processing_state(True) | |
| self.main_area.reset() | |
| # Show first image preview | |
| if settings["files"]: | |
| self.main_area.update_preview(settings["files"][0]) | |
| thread = threading.Thread(target=self._process_worker, args=(settings,), daemon=True) | |
| thread.start() | |
| def _process_worker(self, settings): | |
| """Background processing worker.""" | |
| try: | |
| self.log(f"Starting processing: {len(settings['files'])} file(s)", "info") | |
| self.log(f"Settings: DPI={settings['dpi']}, Lang={settings['lang']}", "info") | |
| t0 = time.time() | |
| def progress_callback(current, total, msg): | |
| self.main_area.update_progress(current, total, msg) | |
| self.log(f"Page {current}/{total}: {msg}", "info") | |
| processor = BatchProcessor( | |
| lang=settings["lang"], | |
| dpi=settings["dpi"], | |
| title=settings["title"], | |
| progress_callback=progress_callback | |
| ) | |
| # Suppress print output, capture via log | |
| import io | |
| old_stdout = sys.stdout | |
| sys.stdout = io.StringIO() | |
| ok = processor.process_batch( | |
| settings["files"], | |
| settings["output"], | |
| title=settings["title"] | |
| ) | |
| sys.stdout = old_stdout | |
| # Calculate quality score | |
| if processor.stats: | |
| ac = sum(s.overall_confidence for s in processor.stats) / len(processor.stats) | |
| tb = sum(s.text_blocks for s in processor.stats) | |
| tt = sum(s.tables for s in processor.stats) | |
| tl = sum(s.lines for s in processor.stats) | |
| tw = sum(s.total_words for s in processor.stats) | |
| lc = sum(s.low_confidence_words for s in processor.stats) | |
| agg_stats = { | |
| 'tables_found': tt, 'confidence': ac, 'blocks': tb, | |
| 'low_conf_words': lc, 'lines_found': tl, | |
| 'columns': max((s.columns for s in processor.stats), default=1), | |
| 'word_count': tw, | |
| } | |
| quality = QualityScorer.calculate(agg_stats) | |
| self.main_area.update_quality(quality) | |
| self.log(f"Quality Score: {quality}/10", "success") | |
| self.log(f"Blocks: {tb}, Tables: {tt}, Words: {tw}, Confidence: {ac:.1f}%", "info") | |
| elapsed = time.time() - t0 | |
| if ok: | |
| self.log(f"Done! Output saved to: {settings['output']}", "success") | |
| self.log(f"Total time: {elapsed:.1f}s", "info") | |
| self.main_area.update_progress(len(settings["files"]), len(settings["files"])) | |
| else: | |
| self.log("Processing failed. Check the log for errors.", "error") | |
| except Exception as e: | |
| self.log(f"Processing error: {e}", "error") | |
| self.log(traceback.format_exc(), "error") | |
| sys.stdout = sys.__stdout__ | |
| finally: | |
| self.processing = False | |
| try: | |
| self.after(0, lambda: self.sidebar.set_processing_state(False)) | |
| except Exception: | |
| pass | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Entry Point | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def main(): | |
| """Launch the application.""" | |
| app = Scan2DocApp() | |
| app.mainloop() | |
| if __name__ == "__main__": | |
| main() | |