import os from langchain_text_splitters import RecursiveCharacterTextSplitter import pytesseract from pdf2image import convert_from_path from dotenv import load_dotenv load_dotenv() TESSERACT_PATH = os.getenv("TESSERACT_PATH", "tesseract") POPPLER_PATH = os.getenv("POPPLER_PATH") pytesseract.pytesseract.tesseract_cmd = TESSERACT_PATH # if not TESSERACT_PATH or not POPPLER_PATH: # raise EnvironmentError( # "TESSERACT_PATH and POPPLER_PATH must be set in your .env file. " # "See README.md for setup instructions." # ) # pytesseract.pytesseract.tesseract_cmd = TESSERACT_PATH def chunk_text(raw_text): """ Splits clean Arabic text into overlapping chunks ready for embedding. Used for both the OCR output and any plain .txt input. """ splitter = RecursiveCharacterTextSplitter( chunk_size=500, chunk_overlap=100, separators=["\n\n", "\n", ".", "؟", "،", " ", ""] ) return splitter.split_text(raw_text) def load_and_chunk_pdf_ocr(pdf_path): """ Uses OCR (Optical Character Recognition) instead of text extraction. This reads the PDF visually, page by page, the same way a human eye would - completely bypassing the corrupted embedded text data we found in testing. """ print("Converting PDF pages to images...") if POPPLER_PATH: images = convert_from_path(pdf_path, poppler_path=POPPLER_PATH, dpi=300) else: images = convert_from_path(pdf_path, dpi=300) print(f"Converted {len(images)} pages to images") print("Running OCR on each page... (this takes a while)") full_text = "" for i, image in enumerate(images): print(f" Processing page {i + 1}/{len(images)}...") page_text = pytesseract.image_to_string(image, lang='ara') full_text += page_text + "\n" print("OCR complete") return chunk_text(full_text) def is_text_garbled(text, sample_size=2000): """ Heuristic check for one specific corruption pattern observed during testing (abnormal 'ى' character frequency). NOTE: testing showed this PDF can also produce a DIFFERENT corruption pattern (mirror-reversed text) that this function does NOT catch. Because of that, this function is currently UNUSED in the main upload pipeline - process_uploaded_pdf() always uses OCR instead of relying on this detector. Kept here for reference and potential future use with a more complete detection strategy. """ sample = text[:sample_size] if len(sample.strip()) < 50: return True words = sample.split() if not words: return True alef_maksura_count = sample.count('ى') alef_maksura_ratio = alef_maksura_count / len(sample) if len(sample) > 0 else 0 if alef_maksura_ratio > 0.05: return True return False def process_uploaded_pdf(pdf_path): """ Processes an uploaded PDF using OCR every time. We deliberately do NOT attempt fast text extraction first. Testing showed this specific PDF (and likely others with similar non-standard Arabic font encoding) produces inconsistent corruption patterns across extraction attempts - sometimes character-noise garbling, sometimes mirror-reversed text. A single heuristic detector cannot reliably catch every failure mode, so we prioritize reliability over speed and always use OCR, which has proven correct across all our testing. """ print("Processing PDF with OCR (this ensures reliable Arabic text extraction)...") return load_and_chunk_pdf_ocr(pdf_path)