IDP-Machine-learning / ocr_engine.py
mrrobot2610's picture
Initial commit: IDP (Intelligent Document Processing) System
1a7ee60
Raw History Blame Contribute Delete
9.62 kB
"""
Lightweight OCR Engine using EasyOCR
Optimized for CPU inference with high accuracy
Supports invoices, receipts, and forms
"""
# Fix for TensorFlow/PaddlePaddle mutex warnings on macOS
import os
os.environ['KMP_DUPLICATE_LIB_OK'] = 'TRUE'
os.environ['OMP_NUM_THREADS'] = '1'
os.environ['OPENBLAS_NUM_THREADS'] = '1'
os.environ['MKL_NUM_THREADS'] = '1'
os.environ['VECLIB_MAXIMUM_THREADS'] = '1'
os.environ['NUMEXPR_NUM_THREADS'] = '1'
# Suppress TensorFlow warnings
os.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'
import warnings
warnings.filterwarnings('ignore', category=UserWarning)
warnings.filterwarnings('ignore', category=FutureWarning)
import numpy as np
from typing import List, Dict, Tuple, Optional
import logging
import easyocr
import json
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
class LightweightOCR:
"""
Lightweight OCR engine based on EasyOCR
Optimized for CPU inference and document processing
"""
def __init__(
self,
lang: str = 'en',
use_gpu: bool = False,
use_angle_cls: bool = True
):
"""
Initialize EasyOCR engine
Args:
lang: Language code ('en' for English)
use_gpu: Use GPU if available (False for CPU-only deployment)
use_angle_cls: Not used in EasyOCR (kept for compatibility)
"""
logger.info(f"Initializing EasyOCR (language: {lang}, GPU: {use_gpu})")
try:
# Initialize EasyOCR Reader
# EasyOCR supports multiple languages, here we use English
self.reader = easyocr.Reader(['en'], gpu=use_gpu)
logger.info("EasyOCR initialized successfully")
except Exception as e:
logger.error(f"EasyOCR initialization failed: {str(e)}")
raise
def extract_text(
self,
image: np.ndarray,
return_boxes: bool = True,
confidence_threshold: float = 0.5
) -> Dict:
"""
Extract text from document image
Args:
image: Input image as numpy array (BGR or RGB)
return_boxes: Include bounding boxes in output
confidence_threshold: Minimum confidence score to include results
Returns:
Dictionary containing:
- text: Full extracted text
- boxes: List of word/line detections with boxes and confidence
- lines: Grouped by lines
"""
logger.info("Starting OCR extraction")
# Run EasyOCR
# detail=1 returns [box, text, confidence]
results = self.reader.readtext(image, detail=1)
if not results:
logger.warning("No text detected in image")
return {
"text": "",
"boxes": [],
"lines": []
}
# Parse results
full_text_parts = []
boxes = []
lines = []
for idx, (box_coords, text, confidence) in enumerate(results):
# Filter by confidence
if confidence < confidence_threshold:
continue
# Add to full text
full_text_parts.append(text)
# Convert box to simple bbox format [x1, y1, x2, y2]
# EasyOCR returns [[x1,y1], [x2,y1], [x2,y2], [x1,y2]]
box_array = np.array(box_coords)
x_coords = box_array[:, 0]
y_coords = box_array[:, 1]
bbox = [
float(np.min(x_coords)),
float(np.min(y_coords)),
float(np.max(x_coords)),
float(np.max(y_coords))
]
# Create box entry
box_entry = {
"text": text,
"bbox": bbox,
"confidence": float(confidence),
"line_number": idx
}
boxes.append(box_entry)
# Create line entry
line_entry = {
"text": text,
"bbox": bbox,
"confidence": float(confidence)
}
lines.append(line_entry)
# Combine full text
full_text = "\n".join(full_text_parts)
result_dict = {
"text": full_text,
"boxes": boxes if return_boxes else [],
"lines": lines
}
logger.info(f"Extracted {len(lines)} lines with avg confidence: "
f"{np.mean([l['confidence'] for l in lines]) if lines else 0:.3f}")
return result_dict
def get_text_with_positions(
self,
image: np.ndarray,
confidence_threshold: float = 0.5
) -> Tuple[str, List[Dict]]:
"""
Convenience method to get both text and position information
Args:
image: Input image
confidence_threshold: Minimum confidence
Returns:
Tuple of (full_text, list of box dictionaries)
"""
result = self.extract_text(image, return_boxes=True, confidence_threshold=confidence_threshold)
return result["text"], result["boxes"]
class PDFtoImageConverter:
"""
Convert PDF pages to images for OCR processing
Lightweight implementation
"""
@staticmethod
def pdf_to_images(pdf_path: str, dpi: int = 200) -> List[np.ndarray]:
"""
Convert PDF to list of images
Args:
pdf_path: Path to PDF file
dpi: Resolution for conversion (200 is good for OCR)
Returns:
List of images as numpy arrays
"""
try:
from pdf2image import convert_from_path
import cv2
logger.info(f"Converting PDF to images: {pdf_path}")
# Convert PDF to PIL images
pil_images = convert_from_path(pdf_path, dpi=dpi)
# Convert to numpy arrays (OpenCV format)
images = []
for pil_img in pil_images:
# Convert PIL RGB to OpenCV BGR
img_array = np.array(pil_img)
img_bgr = cv2.cvtColor(img_array, cv2.COLOR_RGB2BGR)
images.append(img_bgr)
logger.info(f"Converted {len(images)} pages")
return images
except ImportError:
logger.error("pdf2image not installed. Install with: pip install pdf2image")
logger.error("Also requires poppler-utils system package")
raise
except Exception as e:
logger.error(f"Error converting PDF: {str(e)}")
raise
def extract_text_from_file(
file_path: str,
lang: str = 'en',
use_gpu: bool = False,
confidence_threshold: float = 0.5
) -> Dict:
"""
Convenience function to extract text from image or PDF file
Args:
file_path: Path to image or PDF file
lang: Language code
use_gpu: Use GPU for OCR
confidence_threshold: Minimum confidence score
Returns:
Dictionary with extracted text and metadata
"""
import cv2
import os
# Initialize OCR
ocr_engine = LightweightOCR(lang=lang, use_gpu=use_gpu)
# Check file type
ext = os.path.splitext(file_path)[1].lower()
if ext == '.pdf':
# Convert PDF to images
images = PDFtoImageConverter.pdf_to_images(file_path)
# Process each page
results = []
for page_num, image in enumerate(images):
logger.info(f"Processing page {page_num + 1}/{len(images)}")
page_result = ocr_engine.extract_text(image, confidence_threshold=confidence_threshold)
page_result['page_number'] = page_num + 1
results.append(page_result)
return {
"file_path": file_path,
"file_type": "pdf",
"num_pages": len(images),
"pages": results
}
else:
# Load image
image = cv2.imread(file_path)
if image is None:
raise ValueError(f"Could not load image from {file_path}")
# Process single image
result = ocr_engine.extract_text(image, confidence_threshold=confidence_threshold)
return {
"file_path": file_path,
"file_type": "image",
"num_pages": 1,
"pages": [result]
}
if __name__ == "__main__":
import sys
if len(sys.argv) < 2:
print("Usage: python ocr_engine.py <image_or_pdf_path>")
sys.exit(1)
input_path = sys.argv[1]
output_path = "ocr_output.json"
# Extract text
result = extract_text_from_file(input_path)
# Save to JSON
with open(output_path, 'w', encoding='utf-8') as f:
json.dump(result, f, indent=2, ensure_ascii=False)
print(f"\nOCR results saved to {output_path}")
print(f"\nExtracted text preview:")
print("-" * 50)
for page in result['pages']:
print(page['text'][:500]) # First 500 chars
if len(page['text']) > 500:
print("...")