Document Question Answering
Transformers
PyTorch
English
document-processing
ocr
ner
text-classification
information-extraction
invoice
receipt
form
Instructions to use mrrobot2610/IDP-Machine-learning with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use mrrobot2610/IDP-Machine-learning with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("document-question-answering", model="mrrobot2610/IDP-Machine-learning")# pip install -U transformers accelerate # Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("mrrobot2610/IDP-Machine-learning", device_map="auto") - Notebooks
- Google Colab
- Kaggle
File size: 9,801 Bytes
1a7ee60 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 | """
Lightweight Image Preprocessing Pipeline for IDP
Uses OpenCV and Pillow for CPU-friendly operations
Optimizes document images for OCR quality
"""
import cv2
import numpy as np
from PIL import Image, ImageEnhance
from typing import Tuple, Optional
import logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
class DocumentPreprocessor:
"""
Lightweight document preprocessing pipeline optimized for OCR
All operations are CPU-friendly and designed for speed
"""
def __init__(
self,
max_width: int = 2048,
max_height: int = 2048,
enable_deskew: bool = True,
enable_denoise: bool = True,
enable_contrast: bool = True,
):
"""
Args:
max_width: Maximum width for resizing
max_height: Maximum height for resizing
enable_deskew: Enable deskewing correction
enable_denoise: Enable noise reduction
enable_contrast: Enable contrast enhancement
"""
self.max_width = max_width
self.max_height = max_height
self.enable_deskew = enable_deskew
self.enable_denoise = enable_denoise
self.enable_contrast = enable_contrast
def preprocess(
self,
image: np.ndarray,
adaptive_threshold: bool = False
) -> np.ndarray:
"""
Complete preprocessing pipeline
Args:
image: Input image as numpy array (BGR or RGB)
adaptive_threshold: Apply adaptive thresholding for poor quality scans
Returns:
Preprocessed image ready for OCR
"""
logger.info("Starting preprocessing pipeline")
# Step 1: Auto-rotation (from EXIF metadata)
image = self._auto_rotate(image)
# Step 2: Resize to optimal dimensions
image = self._resize_image(image)
# Step 3: Deskew correction
if self.enable_deskew:
image = self._deskew_image(image)
# Step 4: Denoise
if self.enable_denoise:
image = self._denoise_image(image)
# Step 5: Contrast enhancement
if self.enable_contrast:
image = self._enhance_contrast(image)
# Step 6: Adaptive thresholding (optional, for very poor scans)
if adaptive_threshold:
image = self._adaptive_threshold(image)
logger.info("Preprocessing complete")
return image
def _auto_rotate(self, image: np.ndarray) -> np.ndarray:
"""
Auto-rotate image based on EXIF orientation
For images without EXIF, uses simple heuristics
"""
# Convert to PIL to read EXIF
if len(image.shape) == 2:
pil_image = Image.fromarray(image)
else:
# OpenCV uses BGR, PIL uses RGB
rgb_image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)
pil_image = Image.fromarray(rgb_image)
# Try to get EXIF orientation
try:
exif = pil_image._getexif()
if exif:
orientation = exif.get(274) # 274 is orientation tag
if orientation == 3:
pil_image = pil_image.rotate(180, expand=True)
elif orientation == 6:
pil_image = pil_image.rotate(270, expand=True)
elif orientation == 8:
pil_image = pil_image.rotate(90, expand=True)
except (AttributeError, KeyError, TypeError):
# No EXIF data, skip auto-rotation
pass
# Convert back to numpy
image = np.array(pil_image)
if len(image.shape) == 3:
image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)
return image
def _resize_image(self, image: np.ndarray) -> np.ndarray:
"""
Resize image to optimal dimensions for OCR
Maintains aspect ratio
"""
h, w = image.shape[:2]
# Calculate scaling factor
scale = min(self.max_width / w, self.max_height / h, 1.0)
if scale < 1.0:
new_w = int(w * scale)
new_h = int(h * scale)
image = cv2.resize(image, (new_w, new_h), interpolation=cv2.INTER_AREA)
logger.info(f"Resized from ({w}, {h}) to ({new_w}, {new_h})")
return image
def _deskew_image(self, image: np.ndarray) -> np.ndarray:
"""
Detect and correct skew using Hough line transform
Fast and efficient for typical document skews
"""
# Convert to grayscale if needed
if len(image.shape) == 3:
gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
else:
gray = image.copy()
# Edge detection
edges = cv2.Canny(gray, 50, 150, apertureSize=3)
# Detect lines
lines = cv2.HoughLines(edges, 1, np.pi / 180, 200)
if lines is not None and len(lines) > 0:
# Calculate angles
angles = []
for rho, theta in lines[:, 0]:
angle = np.degrees(theta) - 90
if -45 < angle < 45: # Only consider reasonable skew angles
angles.append(angle)
if angles:
# Median angle is most robust
skew_angle = np.median(angles)
# Only correct if skew is significant (> 0.5 degrees)
if abs(skew_angle) > 0.5:
logger.info(f"Detected skew: {skew_angle:.2f} degrees")
# Rotate image
h, w = image.shape[:2]
center = (w // 2, h // 2)
M = cv2.getRotationMatrix2D(center, skew_angle, 1.0)
image = cv2.warpAffine(
image, M, (w, h),
flags=cv2.INTER_CUBIC,
borderMode=cv2.BORDER_REPLICATE
)
return image
def _denoise_image(self, image: np.ndarray) -> np.ndarray:
"""
Apply bilateral filter for noise reduction
Preserves edges while smoothing noise
"""
# Bilateral filter: smooths noise but preserves edges
# d: diameter of pixel neighborhood
# sigmaColor: filter sigma in color space
# sigmaSpace: filter sigma in coordinate space
denoised = cv2.bilateralFilter(image, d=5, sigmaColor=50, sigmaSpace=50)
logger.info("Applied bilateral denoising")
return denoised
def _enhance_contrast(self, image: np.ndarray) -> np.ndarray:
"""
Enhance contrast using CLAHE (Contrast Limited Adaptive Histogram Equalization)
More effective than global histogram equalization for documents
"""
# Convert to LAB color space for better contrast adjustment
if len(image.shape) == 3:
lab = cv2.cvtColor(image, cv2.COLOR_BGR2LAB)
l, a, b = cv2.split(lab)
else:
l = image.copy()
# Apply CLAHE to L channel
clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))
l = clahe.apply(l)
# Merge back
if len(image.shape) == 3:
lab = cv2.merge([l, a, b])
image = cv2.cvtColor(lab, cv2.COLOR_LAB2BGR)
else:
image = l
logger.info("Applied CLAHE contrast enhancement")
return image
def _adaptive_threshold(self, image: np.ndarray) -> np.ndarray:
"""
Apply adaptive thresholding for poor quality scans
Converts to binary image
"""
# Convert to grayscale if needed
if len(image.shape) == 3:
gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
else:
gray = image.copy()
# Adaptive threshold
binary = cv2.adaptiveThreshold(
gray,
255,
cv2.ADAPTIVE_THRESH_GAUSSIAN_C,
cv2.THRESH_BINARY,
blockSize=11,
C=2
)
logger.info("Applied adaptive thresholding")
return binary
def preprocess_for_ocr(
image_path: str,
max_width: int = 2048,
adaptive_threshold: bool = False
) -> np.ndarray:
"""
Convenience function to preprocess an image file for OCR
Args:
image_path: Path to input image
max_width: Maximum width for resizing
adaptive_threshold: Apply adaptive thresholding
Returns:
Preprocessed image as numpy array
"""
# Load image
image = cv2.imread(image_path)
if image is None:
raise ValueError(f"Could not load image from {image_path}")
# Preprocess
preprocessor = DocumentPreprocessor(max_width=max_width)
processed = preprocessor.preprocess(image, adaptive_threshold=adaptive_threshold)
return processed
if __name__ == "__main__":
# Example usage
import sys
if len(sys.argv) < 2:
print("Usage: python preprocessing.py <image_path>")
sys.exit(1)
input_path = sys.argv[1]
output_path = "preprocessed_output.png"
# Preprocess
processed = preprocess_for_ocr(input_path)
# Save result
cv2.imwrite(output_path, processed)
print(f"Preprocessed image saved to {output_path}")
|