Dbmaxwell's picture
chore: remove LightOnOCR integration completely
23b8566 verified
Raw
History Blame Contribute Delete
5.5 kB
from __future__ import annotations
import hashlib
from collections import Counter
from pathlib import Path
import pymupdf
from .models import (
ExtractedDocument,
ExtractedPage,
ExtractionMethod,
WordBox,
)
class ExtractionError(RuntimeError):
"""Raised when a document cannot be converted into positioned words."""
SUPPORTED_SUFFIXES = {".pdf", ".png", ".jpg", ".jpeg", ".tif", ".tiff"}
def _sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def _word_boxes(
raw_words: list[tuple],
page_number: int,
page_width: float,
page_height: float,
method: ExtractionMethod,
orientation_degrees: int,
) -> list[WordBox]:
if orientation_degrees in {90, 270}:
normalized_width = page_height
normalized_height = page_width
else:
normalized_width = page_width
normalized_height = page_height
x_scale = 950.0 / normalized_width
y_scale = 1200.0 / normalized_height
confidence = 1.0 if method == ExtractionMethod.NATIVE_PDF else 0.78
boxes: list[WordBox] = []
for item in raw_words:
text = str(item[4]).strip()
if not text:
continue
x0, y0, x1, y1 = map(float, item[:4])
if orientation_degrees == 90:
x0, y0, x1, y1 = page_height - y1, x0, page_height - y0, x1
elif orientation_degrees == 180:
x0, y0, x1, y1 = (
page_width - x1,
page_height - y1,
page_width - x0,
page_height - y0,
)
elif orientation_degrees == 270:
x0, y0, x1, y1 = y0, page_width - x1, y1, page_width - x0
boxes.append(
WordBox(
text=text,
page=page_number,
x0=x0 * x_scale,
y0=y0 * y_scale,
x1=x1 * x_scale,
y1=y1 * y_scale,
method=method,
confidence=confidence,
)
)
return sorted(boxes, key=lambda word: (word.y0, word.x0))
def _dominant_orientation(page, text_page=None) -> int:
text_dict = page.get_text("dict", textpage=text_page)
directions: Counter[tuple[float, float]] = Counter()
for block in text_dict.get("blocks", []):
for line in block.get("lines", []):
direction = tuple(round(float(value), 1) for value in line.get("dir", (1, 0)))
weight = sum(len(span.get("text", "")) for span in line.get("spans", []))
directions[direction] += weight
if not directions:
return 0
direction, _ = directions.most_common(1)[0]
return {
(1.0, 0.0): 0,
(0.0, -1.0): 90,
(-1.0, 0.0): 180,
(0.0, 1.0): 270,
}.get(direction, 0)
def extract_document(path: str | Path) -> ExtractedDocument:
source_path = Path(path).expanduser().resolve()
if not source_path.is_file():
raise ExtractionError(f"Dosya bulunamadı: {source_path}")
if source_path.suffix.lower() not in SUPPORTED_SUFFIXES:
raise ExtractionError(
"Desteklenmeyen dosya türü. PDF, PNG, JPG veya TIFF yükleyin."
)
try:
document = pymupdf.open(source_path)
except Exception as exc: # pragma: no cover - delegated library failures
raise ExtractionError(f"Belge açılamadı: {exc}") from exc
pages: list[ExtractedPage] = []
for page_index, page in enumerate(document):
native_words = page.get_text("words", sort=True)
native_text_length = sum(len(str(word[4])) for word in native_words)
if native_text_length >= 40:
method = ExtractionMethod.NATIVE_PDF
raw_words = native_words
text_page = None
else:
try:
text_page = page.get_textpage_ocr(
language="tur+eng",
dpi=300,
full=True,
)
raw_words = page.get_text("words", textpage=text_page, sort=True)
method = ExtractionMethod.OCR
except Exception as exc:
raise ExtractionError(
"Belge metin katmanı içermiyor ve OCR başlatılamadı. "
"Tesseract ile Türkçe/İngilizce dil paketlerinin kurulu "
f"olduğunu doğrulayın. Ayrıntı: {exc}"
) from exc
orientation_degrees = _dominant_orientation(page, text_page)
words = _word_boxes(
raw_words=raw_words,
page_number=page_index + 1,
page_width=float(page.rect.width),
page_height=float(page.rect.height),
method=method,
orientation_degrees=orientation_degrees,
)
if not words:
raise ExtractionError(f"{page_index + 1}. sayfadan metin çıkarılamadı.")
pages.append(
ExtractedPage(
number=page_index + 1,
width=950.0,
height=1200.0,
words=words,
method=method,
orientation_degrees=orientation_degrees,
)
)
document.close()
return ExtractedDocument(
source_path=str(source_path),
source_filename=source_path.name,
sha256=_sha256(source_path),
pages=pages,
)