""" Lightweight OCR Engine using EasyOCR Optimized for CPU inference with high accuracy Supports invoices, receipts, and forms """ # Fix for TensorFlow/PaddlePaddle mutex warnings on macOS import os os.environ['KMP_DUPLICATE_LIB_OK'] = 'TRUE' os.environ['OMP_NUM_THREADS'] = '1' os.environ['OPENBLAS_NUM_THREADS'] = '1' os.environ['MKL_NUM_THREADS'] = '1' os.environ['VECLIB_MAXIMUM_THREADS'] = '1' os.environ['NUMEXPR_NUM_THREADS'] = '1' # Suppress TensorFlow warnings os.environ['TF_CPP_MIN_LOG_LEVEL'] = '2' import warnings warnings.filterwarnings('ignore', category=UserWarning) warnings.filterwarnings('ignore', category=FutureWarning) import numpy as np from typing import List, Dict, Tuple, Optional import logging import easyocr import json logging.basicConfig(level=logging.INFO) logger = logging.getLogger(__name__) class LightweightOCR: """ Lightweight OCR engine based on EasyOCR Optimized for CPU inference and document processing """ def __init__( self, lang: str = 'en', use_gpu: bool = False, use_angle_cls: bool = True ): """ Initialize EasyOCR engine Args: lang: Language code ('en' for English) use_gpu: Use GPU if available (False for CPU-only deployment) use_angle_cls: Not used in EasyOCR (kept for compatibility) """ logger.info(f"Initializing EasyOCR (language: {lang}, GPU: {use_gpu})") try: # Initialize EasyOCR Reader # EasyOCR supports multiple languages, here we use English self.reader = easyocr.Reader(['en'], gpu=use_gpu) logger.info("EasyOCR initialized successfully") except Exception as e: logger.error(f"EasyOCR initialization failed: {str(e)}") raise def extract_text( self, image: np.ndarray, return_boxes: bool = True, confidence_threshold: float = 0.5 ) -> Dict: """ Extract text from document image Args: image: Input image as numpy array (BGR or RGB) return_boxes: Include bounding boxes in output confidence_threshold: Minimum confidence score to include results Returns: Dictionary containing: - text: Full extracted text - boxes: List of word/line detections with boxes and confidence - lines: Grouped by lines """ logger.info("Starting OCR extraction") # Run EasyOCR # detail=1 returns [box, text, confidence] results = self.reader.readtext(image, detail=1) if not results: logger.warning("No text detected in image") return { "text": "", "boxes": [], "lines": [] } # Parse results full_text_parts = [] boxes = [] lines = [] for idx, (box_coords, text, confidence) in enumerate(results): # Filter by confidence if confidence < confidence_threshold: continue # Add to full text full_text_parts.append(text) # Convert box to simple bbox format [x1, y1, x2, y2] # EasyOCR returns [[x1,y1], [x2,y1], [x2,y2], [x1,y2]] box_array = np.array(box_coords) x_coords = box_array[:, 0] y_coords = box_array[:, 1] bbox = [ float(np.min(x_coords)), float(np.min(y_coords)), float(np.max(x_coords)), float(np.max(y_coords)) ] # Create box entry box_entry = { "text": text, "bbox": bbox, "confidence": float(confidence), "line_number": idx } boxes.append(box_entry) # Create line entry line_entry = { "text": text, "bbox": bbox, "confidence": float(confidence) } lines.append(line_entry) # Combine full text full_text = "\n".join(full_text_parts) result_dict = { "text": full_text, "boxes": boxes if return_boxes else [], "lines": lines } logger.info(f"Extracted {len(lines)} lines with avg confidence: " f"{np.mean([l['confidence'] for l in lines]) if lines else 0:.3f}") return result_dict def get_text_with_positions( self, image: np.ndarray, confidence_threshold: float = 0.5 ) -> Tuple[str, List[Dict]]: """ Convenience method to get both text and position information Args: image: Input image confidence_threshold: Minimum confidence Returns: Tuple of (full_text, list of box dictionaries) """ result = self.extract_text(image, return_boxes=True, confidence_threshold=confidence_threshold) return result["text"], result["boxes"] class PDFtoImageConverter: """ Convert PDF pages to images for OCR processing Lightweight implementation """ @staticmethod def pdf_to_images(pdf_path: str, dpi: int = 200) -> List[np.ndarray]: """ Convert PDF to list of images Args: pdf_path: Path to PDF file dpi: Resolution for conversion (200 is good for OCR) Returns: List of images as numpy arrays """ try: from pdf2image import convert_from_path import cv2 logger.info(f"Converting PDF to images: {pdf_path}") # Convert PDF to PIL images pil_images = convert_from_path(pdf_path, dpi=dpi) # Convert to numpy arrays (OpenCV format) images = [] for pil_img in pil_images: # Convert PIL RGB to OpenCV BGR img_array = np.array(pil_img) img_bgr = cv2.cvtColor(img_array, cv2.COLOR_RGB2BGR) images.append(img_bgr) logger.info(f"Converted {len(images)} pages") return images except ImportError: logger.error("pdf2image not installed. Install with: pip install pdf2image") logger.error("Also requires poppler-utils system package") raise except Exception as e: logger.error(f"Error converting PDF: {str(e)}") raise def extract_text_from_file( file_path: str, lang: str = 'en', use_gpu: bool = False, confidence_threshold: float = 0.5 ) -> Dict: """ Convenience function to extract text from image or PDF file Args: file_path: Path to image or PDF file lang: Language code use_gpu: Use GPU for OCR confidence_threshold: Minimum confidence score Returns: Dictionary with extracted text and metadata """ import cv2 import os # Initialize OCR ocr_engine = LightweightOCR(lang=lang, use_gpu=use_gpu) # Check file type ext = os.path.splitext(file_path)[1].lower() if ext == '.pdf': # Convert PDF to images images = PDFtoImageConverter.pdf_to_images(file_path) # Process each page results = [] for page_num, image in enumerate(images): logger.info(f"Processing page {page_num + 1}/{len(images)}") page_result = ocr_engine.extract_text(image, confidence_threshold=confidence_threshold) page_result['page_number'] = page_num + 1 results.append(page_result) return { "file_path": file_path, "file_type": "pdf", "num_pages": len(images), "pages": results } else: # Load image image = cv2.imread(file_path) if image is None: raise ValueError(f"Could not load image from {file_path}") # Process single image result = ocr_engine.extract_text(image, confidence_threshold=confidence_threshold) return { "file_path": file_path, "file_type": "image", "num_pages": 1, "pages": [result] } if __name__ == "__main__": import sys if len(sys.argv) < 2: print("Usage: python ocr_engine.py ") sys.exit(1) input_path = sys.argv[1] output_path = "ocr_output.json" # Extract text result = extract_text_from_file(input_path) # Save to JSON with open(output_path, 'w', encoding='utf-8') as f: json.dump(result, f, indent=2, ensure_ascii=False) print(f"\nOCR results saved to {output_path}") print(f"\nExtracted text preview:") print("-" * 50) for page in result['pages']: print(page['text'][:500]) # First 500 chars if len(page['text']) > 500: print("...")