Essay-grader-ai / modules /file_processor.py
Turbiling's picture
Update modules/file_processor.py
bd6eeb8 verified
Raw
History Blame Contribute Delete
8.12 kB
# modules/file_processor.py
import os
import logging
from typing import Optional
from docx import Document
import PyPDF2
from PIL import Image
import pytesseract
from pdf2image import convert_from_path
import cv2
import numpy as np
logger = logging.getLogger(__name__)
class FileProcessor:
def __init__(self):
logger.info("FileProcessor initialized with OCR support")
def extract_text(self, file_path: str) -> str:
if not os.path.exists(file_path):
raise FileNotFoundError(f"File not found: {file_path}")
_, ext = os.path.splitext(file_path)
ext = ext.lower()
logger.info(f"Extracting text from {ext} file")
try:
if ext == '.txt':
return self._extract_from_txt(file_path)
elif ext == '.docx':
return self._extract_from_docx(file_path)
elif ext == '.pdf':
return self._extract_from_pdf(file_path)
else:
raise ValueError(f"Unsupported file type: {ext}")
except Exception as e:
logger.error(f"Error extracting text: {e}")
raise
def _extract_from_txt(self, path: str) -> str:
try:
with open(path, 'r', encoding='utf-8') as f:
text = f.read()
logger.info(f"Extracted {len(text)} chars from TXT")
return text
except UnicodeDecodeError:
with open(path, 'r', encoding='latin-1') as f:
text = f.read()
logger.warning("Used latin-1 encoding for TXT")
return text
def _extract_from_docx(self, path: str) -> str:
doc = Document(path)
paragraphs = []
for para in doc.paragraphs:
text = para.text.strip()
if text:
paragraphs.append(text)
text = '\n\n'.join(paragraphs)
logger.info(f"Extracted {len(text)} chars from DOCX")
return text
def _extract_from_pdf(self, path: str) -> str:
"""Extract text from PDF with OCR fallback"""
# پہلے PyPDF2 سے کوشش کریں
text = self._extract_with_pypdf2(path)
# اگر کم از کم 30 الفاظ نہیں ملے تو OCR استعمال کریں
word_count = len(text.split())
if word_count < 30:
logger.info(f"PyPDF2 extracted only {word_count} words. Trying OCR...")
ocr_text = self._extract_with_ocr(path)
# جو زیادہ text دے وہ استعمال کریں
if len(ocr_text) > len(text):
text = ocr_text
logger.info(f"OCR extraction successful: {len(text)} chars")
else:
logger.warning("OCR did not improve extraction")
logger.info(f"Final extracted text: {len(text)} chars from PDF")
return text
def _extract_with_pypdf2(self, path: str) -> str:
"""Regular PDF text extraction"""
text_parts = []
try:
with open(path, 'rb') as f:
reader = PyPDF2.PdfReader(f)
num_pages = len(reader.pages)
logger.info(f"Reading {num_pages} pages with PyPDF2")
for page_num in range(num_pages):
page = reader.pages[page_num]
text = page.extract_text()
if text.strip():
text_parts.append(text)
return '\n\n'.join(text_parts)
except Exception as e:
logger.error(f"PyPDF2 extraction failed: {e}")
return ""
def _extract_with_ocr(self, path: str) -> str:
"""OCR-based extraction for scanned PDFs"""
text_parts = []
try:
# PDF کو images میں convert کریں
logger.info("Converting PDF to images for OCR...")
images = convert_from_path(path, dpi=300)
logger.info(f"Processing {len(images)} pages with OCR...")
for i, image in enumerate(images):
logger.info(f"OCR on page {i+1}/{len(images)}...")
# Image کو pre-process کریں (better OCR)
processed_image = self._preprocess_image(image)
# OCR - Urdu اور English دونوں
try:
# پہلے English try کریں
text_eng = pytesseract.image_to_string(
processed_image,
lang='eng',
config='--psm 6'
)
# پھر Urdu try کریں
text_urd = pytesseract.image_to_string(
processed_image,
lang='urd',
config='--psm 6'
)
# جو زیادہ text دے وہ استعمال کریں
if len(text_eng) > len(text_urd):
text = text_eng
logger.info(f"Page {i+1}: English OCR - {len(text)} chars")
else:
text = text_urd
logger.info(f"Page {i+1}: Urdu OCR - {len(text)} chars")
if text.strip():
text_parts.append(text.strip())
except Exception as e:
logger.error(f"OCR failed on page {i+1}: {e}")
# Fallback to English only
try:
text = pytesseract.image_to_string(processed_image, lang='eng')
if text.strip():
text_parts.append(text.strip())
except:
logger.error(f"English OCR also failed on page {i+1}")
final_text = '\n\n'.join(text_parts)
logger.info(f"OCR complete: {len(final_text)} chars extracted")
return final_text
except Exception as e:
logger.error(f"OCR extraction failed: {e}")
return ""
def _preprocess_image(self, pil_image):
"""Image کو OCR کے لیے بہتر بنائیں"""
try:
# PIL to numpy array
img = np.array(pil_image)
# Convert to grayscale
if len(img.shape) == 3:
gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)
else:
gray = img
# Increase contrast
alpha = 1.5 # Contrast
beta = 0 # Brightness
adjusted = cv2.convertScaleAbs(gray, alpha=alpha, beta=beta)
# Denoise
denoised = cv2.fastNlMeansDenoising(adjusted, None, 10, 7, 21)
# Threshold
_, thresh = cv2.threshold(
denoised, 0, 255,
cv2.THRESH_BINARY + cv2.THRESH_OTSU
)
# Convert back to PIL
processed = Image.fromarray(thresh)
return processed
except Exception as e:
logger.error(f"Image preprocessing failed: {e}")
return pil_image
def detect_language(self, text: str) -> str:
"""زبان کی شناخت"""
urdu_chars = sum(1 for c in text if '\u0600' <= c <= '\u06FF')
english_chars = sum(1 for c in text if c.isalpha() and c.isascii())
total_chars = urdu_chars + english_chars
if total_chars == 0:
return 'unknown'
urdu_percent = (urdu_chars / total_chars) * 100
if urdu_percent > 70:
return 'urdu'
elif urdu_percent < 30:
return 'english'
else:
return 'mixed'