fyp / preprocess.py
waheed7360's picture
Update preprocess.py
8e987de verified
Raw
History Blame Contribute Delete
3.43 kB
import cv2
import numpy as np
from PIL import Image
def preprocess_image(image, target_width=420, target_height=130):
"""
Preprocess input image to match training data format:
- Convert to black background with white foreground
- Resize to target dimensions
- Enhance contrast for better detection
Args:
image: Input image (numpy array or PIL Image)
target_width: Width to resize to (default: 420)
target_height: Height to resize to (default: 130)
Returns:
Preprocessed image as numpy array
"""
# Convert PIL Image to numpy array if needed
if isinstance(image, Image.Image):
image_np = np.array(image)
else:
image_np = image.copy()
# Convert to grayscale if colored
if len(image_np.shape) == 3 and image_np.shape[2] == 3:
gray_img = cv2.cvtColor(image_np, cv2.COLOR_RGB2GRAY)
else:
gray_img = image_np
# Make sure image is uint8 for cv2 operations
if gray_img.dtype != np.uint8:
gray_img = (gray_img * 255).astype(np.uint8)
# Determine foreground/background colors
mean_value = np.mean(gray_img)
is_dark_background = mean_value < 128
# Invert if needed to get dark background/light foreground
if not is_dark_background:
gray_img = 255 - gray_img
# Apply adaptive thresholding to enhance contrast
# This helps with handwritten content with varying pressure/intensity
binary_img = cv2.adaptiveThreshold(
gray_img,
255,
cv2.ADAPTIVE_THRESH_GAUSSIAN_C,
cv2.THRESH_BINARY_INV,
11, # Block size
2 # Constant subtracted from mean
)
# Resize to target dimensions
# Use INTER_AREA for shrinking, INTER_CUBIC for enlarging
current_height, current_width = binary_img.shape
if current_width > target_width or current_height > target_height:
interpolation = cv2.INTER_AREA
else:
interpolation = cv2.INTER_CUBIC
# Calculate scaling ratio to maintain aspect ratio
width_ratio = target_width / current_width
height_ratio = target_height / current_height
scaling_ratio = min(width_ratio, height_ratio)
# Calculate new dimensions
new_width = int(current_width * scaling_ratio)
new_height = int(current_height * scaling_ratio)
# Resize while maintaining aspect ratio
resized_img = cv2.resize(binary_img, (new_width, new_height), interpolation=interpolation)
# Create black canvas of target size
final_img = np.zeros((target_height, target_width), dtype=np.uint8)
# Calculate position to center the resized image
x_offset = (target_width - new_width) // 2
y_offset = (target_height - new_height) // 2
# Place the resized image on the canvas
final_img[y_offset:y_offset+new_height, x_offset:x_offset+new_width] = resized_img
# Invert to get white foreground on black background
final_img = 255 - final_img
# Apply additional contrast enhancement if needed
# This can help with faint or inconsistent handwriting
clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))
enhanced_img = clahe.apply(final_img)
# Optional: Apply slight Gaussian blur to reduce noise
# Only apply if the image seems noisy
enhanced_img = cv2.GaussianBlur(enhanced_img, (3, 3), 0)
return enhanced_img