Document Question Answering
Transformers
PyTorch
English
document-processing
ocr
ner
text-classification
information-extraction
invoice
receipt
form
Instructions to use mrrobot2610/IDP-Machine-learning with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use mrrobot2610/IDP-Machine-learning with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("document-question-answering", model="mrrobot2610/IDP-Machine-learning")# pip install -U transformers accelerate # Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("mrrobot2610/IDP-Machine-learning", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download ocr_engine.py from mrrobot2610/IDP-Machine-learning: direct link, hf CLI and curl.
- Browser
- Download file 9.62 kB
-
https://huggingface.co/mrrobot2610/IDP-Machine-learning/resolve/main/ocr_engine.py
- Command line
-
hf download hf://mrrobot2610/IDP-Machine-learning/ocr_engine.py
-
curl -L -o ocr_engine.py https://huggingface.co/mrrobot2610/IDP-Machine-learning/resolve/main/ocr_engine.py
9.62 kB
| """ | |
| Lightweight OCR Engine using EasyOCR | |
| Optimized for CPU inference with high accuracy | |
| Supports invoices, receipts, and forms | |
| """ | |
| # Fix for TensorFlow/PaddlePaddle mutex warnings on macOS | |
| import os | |
| os.environ['KMP_DUPLICATE_LIB_OK'] = 'TRUE' | |
| os.environ['OMP_NUM_THREADS'] = '1' | |
| os.environ['OPENBLAS_NUM_THREADS'] = '1' | |
| os.environ['MKL_NUM_THREADS'] = '1' | |
| os.environ['VECLIB_MAXIMUM_THREADS'] = '1' | |
| os.environ['NUMEXPR_NUM_THREADS'] = '1' | |
| # Suppress TensorFlow warnings | |
| os.environ['TF_CPP_MIN_LOG_LEVEL'] = '2' | |
| import warnings | |
| warnings.filterwarnings('ignore', category=UserWarning) | |
| warnings.filterwarnings('ignore', category=FutureWarning) | |
| import numpy as np | |
| from typing import List, Dict, Tuple, Optional | |
| import logging | |
| import easyocr | |
| import json | |
| logging.basicConfig(level=logging.INFO) | |
| logger = logging.getLogger(__name__) | |
| class LightweightOCR: | |
| """ | |
| Lightweight OCR engine based on EasyOCR | |
| Optimized for CPU inference and document processing | |
| """ | |
| def __init__( | |
| self, | |
| lang: str = 'en', | |
| use_gpu: bool = False, | |
| use_angle_cls: bool = True | |
| ): | |
| """ | |
| Initialize EasyOCR engine | |
| Args: | |
| lang: Language code ('en' for English) | |
| use_gpu: Use GPU if available (False for CPU-only deployment) | |
| use_angle_cls: Not used in EasyOCR (kept for compatibility) | |
| """ | |
| logger.info(f"Initializing EasyOCR (language: {lang}, GPU: {use_gpu})") | |
| try: | |
| # Initialize EasyOCR Reader | |
| # EasyOCR supports multiple languages, here we use English | |
| self.reader = easyocr.Reader(['en'], gpu=use_gpu) | |
| logger.info("EasyOCR initialized successfully") | |
| except Exception as e: | |
| logger.error(f"EasyOCR initialization failed: {str(e)}") | |
| raise | |
| def extract_text( | |
| self, | |
| image: np.ndarray, | |
| return_boxes: bool = True, | |
| confidence_threshold: float = 0.5 | |
| ) -> Dict: | |
| """ | |
| Extract text from document image | |
| Args: | |
| image: Input image as numpy array (BGR or RGB) | |
| return_boxes: Include bounding boxes in output | |
| confidence_threshold: Minimum confidence score to include results | |
| Returns: | |
| Dictionary containing: | |
| - text: Full extracted text | |
| - boxes: List of word/line detections with boxes and confidence | |
| - lines: Grouped by lines | |
| """ | |
| logger.info("Starting OCR extraction") | |
| # Run EasyOCR | |
| # detail=1 returns [box, text, confidence] | |
| results = self.reader.readtext(image, detail=1) | |
| if not results: | |
| logger.warning("No text detected in image") | |
| return { | |
| "text": "", | |
| "boxes": [], | |
| "lines": [] | |
| } | |
| # Parse results | |
| full_text_parts = [] | |
| boxes = [] | |
| lines = [] | |
| for idx, (box_coords, text, confidence) in enumerate(results): | |
| # Filter by confidence | |
| if confidence < confidence_threshold: | |
| continue | |
| # Add to full text | |
| full_text_parts.append(text) | |
| # Convert box to simple bbox format [x1, y1, x2, y2] | |
| # EasyOCR returns [[x1,y1], [x2,y1], [x2,y2], [x1,y2]] | |
| box_array = np.array(box_coords) | |
| x_coords = box_array[:, 0] | |
| y_coords = box_array[:, 1] | |
| bbox = [ | |
| float(np.min(x_coords)), | |
| float(np.min(y_coords)), | |
| float(np.max(x_coords)), | |
| float(np.max(y_coords)) | |
| ] | |
| # Create box entry | |
| box_entry = { | |
| "text": text, | |
| "bbox": bbox, | |
| "confidence": float(confidence), | |
| "line_number": idx | |
| } | |
| boxes.append(box_entry) | |
| # Create line entry | |
| line_entry = { | |
| "text": text, | |
| "bbox": bbox, | |
| "confidence": float(confidence) | |
| } | |
| lines.append(line_entry) | |
| # Combine full text | |
| full_text = "\n".join(full_text_parts) | |
| result_dict = { | |
| "text": full_text, | |
| "boxes": boxes if return_boxes else [], | |
| "lines": lines | |
| } | |
| logger.info(f"Extracted {len(lines)} lines with avg confidence: " | |
| f"{np.mean([l['confidence'] for l in lines]) if lines else 0:.3f}") | |
| return result_dict | |
| def get_text_with_positions( | |
| self, | |
| image: np.ndarray, | |
| confidence_threshold: float = 0.5 | |
| ) -> Tuple[str, List[Dict]]: | |
| """ | |
| Convenience method to get both text and position information | |
| Args: | |
| image: Input image | |
| confidence_threshold: Minimum confidence | |
| Returns: | |
| Tuple of (full_text, list of box dictionaries) | |
| """ | |
| result = self.extract_text(image, return_boxes=True, confidence_threshold=confidence_threshold) | |
| return result["text"], result["boxes"] | |
| class PDFtoImageConverter: | |
| """ | |
| Convert PDF pages to images for OCR processing | |
| Lightweight implementation | |
| """ | |
| def pdf_to_images(pdf_path: str, dpi: int = 200) -> List[np.ndarray]: | |
| """ | |
| Convert PDF to list of images | |
| Args: | |
| pdf_path: Path to PDF file | |
| dpi: Resolution for conversion (200 is good for OCR) | |
| Returns: | |
| List of images as numpy arrays | |
| """ | |
| try: | |
| from pdf2image import convert_from_path | |
| import cv2 | |
| logger.info(f"Converting PDF to images: {pdf_path}") | |
| # Convert PDF to PIL images | |
| pil_images = convert_from_path(pdf_path, dpi=dpi) | |
| # Convert to numpy arrays (OpenCV format) | |
| images = [] | |
| for pil_img in pil_images: | |
| # Convert PIL RGB to OpenCV BGR | |
| img_array = np.array(pil_img) | |
| img_bgr = cv2.cvtColor(img_array, cv2.COLOR_RGB2BGR) | |
| images.append(img_bgr) | |
| logger.info(f"Converted {len(images)} pages") | |
| return images | |
| except ImportError: | |
| logger.error("pdf2image not installed. Install with: pip install pdf2image") | |
| logger.error("Also requires poppler-utils system package") | |
| raise | |
| except Exception as e: | |
| logger.error(f"Error converting PDF: {str(e)}") | |
| raise | |
| def extract_text_from_file( | |
| file_path: str, | |
| lang: str = 'en', | |
| use_gpu: bool = False, | |
| confidence_threshold: float = 0.5 | |
| ) -> Dict: | |
| """ | |
| Convenience function to extract text from image or PDF file | |
| Args: | |
| file_path: Path to image or PDF file | |
| lang: Language code | |
| use_gpu: Use GPU for OCR | |
| confidence_threshold: Minimum confidence score | |
| Returns: | |
| Dictionary with extracted text and metadata | |
| """ | |
| import cv2 | |
| import os | |
| # Initialize OCR | |
| ocr_engine = LightweightOCR(lang=lang, use_gpu=use_gpu) | |
| # Check file type | |
| ext = os.path.splitext(file_path)[1].lower() | |
| if ext == '.pdf': | |
| # Convert PDF to images | |
| images = PDFtoImageConverter.pdf_to_images(file_path) | |
| # Process each page | |
| results = [] | |
| for page_num, image in enumerate(images): | |
| logger.info(f"Processing page {page_num + 1}/{len(images)}") | |
| page_result = ocr_engine.extract_text(image, confidence_threshold=confidence_threshold) | |
| page_result['page_number'] = page_num + 1 | |
| results.append(page_result) | |
| return { | |
| "file_path": file_path, | |
| "file_type": "pdf", | |
| "num_pages": len(images), | |
| "pages": results | |
| } | |
| else: | |
| # Load image | |
| image = cv2.imread(file_path) | |
| if image is None: | |
| raise ValueError(f"Could not load image from {file_path}") | |
| # Process single image | |
| result = ocr_engine.extract_text(image, confidence_threshold=confidence_threshold) | |
| return { | |
| "file_path": file_path, | |
| "file_type": "image", | |
| "num_pages": 1, | |
| "pages": [result] | |
| } | |
| if __name__ == "__main__": | |
| import sys | |
| if len(sys.argv) < 2: | |
| print("Usage: python ocr_engine.py <image_or_pdf_path>") | |
| sys.exit(1) | |
| input_path = sys.argv[1] | |
| output_path = "ocr_output.json" | |
| # Extract text | |
| result = extract_text_from_file(input_path) | |
| # Save to JSON | |
| with open(output_path, 'w', encoding='utf-8') as f: | |
| json.dump(result, f, indent=2, ensure_ascii=False) | |
| print(f"\nOCR results saved to {output_path}") | |
| print(f"\nExtracted text preview:") | |
| print("-" * 50) | |
| for page in result['pages']: | |
| print(page['text'][:500]) # First 500 chars | |
| if len(page['text']) > 500: | |
| print("...") | |