Spaces:
Runtime error
Runtime error
File size: 9,806 Bytes
b9d34d8 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 | """PaddleOCR engine wrapper.
Caches one PaddleOCR instance per ``(lang, use_angle_cls)`` key and reuses it.
All OCR predict calls are serialized with a threading.Lock because PaddleOCR's
predictors are not safe to call concurrently from multiple threads.
Targets the PaddleOCR 2.7.3 API:
PaddleOCR(use_angle_cls=<bool>, lang=<code>, show_log=False)
result = instance.ocr(img_bgr, cls=use_angle_cls)
Result structure (PaddleOCR 2.7.x):
[
[
[box, (text, conf)],
...
]
]
where ``result[0]`` may be ``None`` when no text is detected.
"""
import threading
import numpy as np
class OCREngine:
"""Singleton-style PaddleOCR wrapper with per-(lang, angle) caching."""
def __init__(self) -> None:
self._instances = {}
self._instances_lock = threading.Lock()
# Serializes all .ocr() predict calls across threads.
self._predict_lock = threading.Lock()
# ------------------------------------------------------------------
# Instance management
# ------------------------------------------------------------------
def _get_instance(self, lang: str, use_angle_cls: bool):
"""Return a cached PaddleOCR instance, creating it on first use."""
key = (lang, bool(use_angle_cls))
# Fast path: already created.
instance = self._instances.get(key)
if instance is not None:
return instance
with self._instances_lock:
# Re-check inside the lock to avoid double construction.
instance = self._instances.get(key)
if instance is None:
# Imported lazily so importing this module is cheap and does
# not trigger PaddleOCR/Paddle initialization at import time.
from paddleocr import PaddleOCR
# Accuracy-oriented inference tuning (all valid PaddleOCR
# 2.7.3 args, no extra model downloads):
# - det_limit_side_len 1536 (vs default 960): lets the
# detector use our high-DPI render instead of shrinking it,
# so small/dense text is found.
# - det_db_unclip_ratio 1.8 (vs 1.5): expands detected boxes
# so characters at box edges aren't clipped before recog.
# - det_db_box_thresh 0.5 (vs 0.6): recovers fainter text.
# - use_dilation: connects broken strokes in noisy scans.
instance = PaddleOCR(
use_angle_cls=bool(use_angle_cls),
lang=lang,
show_log=False,
det_limit_side_len=1536,
det_limit_type="max",
det_db_unclip_ratio=1.8,
det_db_box_thresh=0.5,
use_dilation=True,
)
self._instances[key] = instance
return instance
def detect_boxes(self, img_bgr: np.ndarray, *, lang: str = "en") -> list:
"""Return text-line boxes (4-point polygons) for the image.
Used by the handwriting pipeline, which detects lines with PaddleOCR but
recognizes them with a handwriting model. We reuse the normal OCR call
and keep only its boxes — PaddleOCR 2.7.3's detection-only (``rec=False``)
path has a numpy truth-value bug, so we avoid it.
"""
lines = self.ocr_lines(img_bgr, lang=lang, use_angle_cls=False)
return [line["box"] for line in lines]
def warmup(self, lang: str = "en") -> None:
"""Preload a PaddleOCR instance (downloads weights on first run).
Called at application startup so the first real request does not pay
the model-load / weight-download cost. Runs a tiny dummy inference to
force lazy predictor initialization.
"""
try:
# Construction (which on first run downloads model weights from the
# network) is the failure-prone step, so it MUST be inside the guard:
# a first-run download failure / corrupt model cache must not abort
# server startup. Lazy construction then retries on the first real
# OCR request, where jobs.py turns any failure into a per-page error.
instance = self._get_instance(lang, use_angle_cls=False)
dummy = np.full((32, 32, 3), 255, dtype=np.uint8)
with self._predict_lock:
instance.ocr(dummy, cls=False)
except Exception:
# Warmup is best-effort; a failure here must not crash startup.
pass
# ------------------------------------------------------------------
# OCR
# ------------------------------------------------------------------
def _run(self, img_bgr: np.ndarray, lang: str, use_angle_cls: bool):
"""Run PaddleOCR under the predict lock and return the raw lines list.
Returns the list of ``[box, (text, conf)]`` entries (possibly empty).
"""
instance = self._get_instance(lang, use_angle_cls)
with self._predict_lock:
result = instance.ocr(img_bgr, cls=bool(use_angle_cls))
if not result:
return []
page = result[0]
if page is None:
return []
return page
@staticmethod
def _line_y(box) -> float:
"""Top-y coordinate of a detection box (4 [x, y] points)."""
return min(float(pt[1]) for pt in box)
@staticmethod
def _line_x(box) -> float:
"""Left-x coordinate of a detection box (4 [x, y] points)."""
return min(float(pt[0]) for pt in box)
@staticmethod
def _line_height(box) -> float:
ys = [float(pt[1]) for pt in box]
return max(ys) - min(ys)
def ocr_lines(
self, img_bgr: np.ndarray, *, lang: str, use_angle_cls: bool
) -> list:
"""Return detected text lines in reading order.
Each element is ``{"text": str, "confidence": float, "box": list}``.
Lines are grouped into rows by their vertical position, and within a
row sorted left-to-right, approximating natural reading order.
NOTE: this assumes a SINGLE-column layout. On a multi-column scan,
side-by-side lines at the same vertical position are merged into one row
and emitted left-to-right, so the two columns come out interleaved. Real
column reconstruction (XY-cut / gutter detection) is not implemented yet.
Born-digital multi-column PDFs take the text-layer path (textlayer.py,
same single-column caveat) or the Gemini path.
"""
raw = self._run(img_bgr, lang, use_angle_cls)
items = []
for entry in raw:
try:
box, payload = entry[0], entry[1]
text, conf = payload[0], payload[1]
if text is None or not box:
continue
# Compute geometry inside the guard so a single malformed
# detection box is skipped rather than aborting the whole page.
y = self._line_y(box)
x = self._line_x(box)
h = self._line_height(box)
except (TypeError, IndexError, ValueError):
continue
items.append(
{
"text": str(text),
"confidence": float(conf),
"box": box,
"_y": y,
"_x": x,
"_h": h,
}
)
if not items:
return []
# Group items into rows: two items share a row if their top-y values
# are within a tolerance derived from the median glyph height.
heights = [it["_h"] for it in items if it["_h"] > 0]
median_h = float(np.median(heights)) if heights else 12.0
tol = max(median_h * 0.6, 6.0)
# Sort primarily by y so we can sweep rows top-to-bottom.
items.sort(key=lambda it: (it["_y"], it["_x"]))
rows = []
current = [items[0]]
current_y = items[0]["_y"]
for it in items[1:]:
if abs(it["_y"] - current_y) <= tol:
current.append(it)
else:
rows.append(current)
current = [it]
current_y = it["_y"]
rows.append(current)
ordered = []
for row in rows:
row.sort(key=lambda it: it["_x"])
for it in row:
ordered.append(
{
"text": it["text"],
"confidence": it["confidence"],
"box": it["box"],
}
)
return ordered
def ocr_text(
self, img_bgr: np.ndarray, *, lang: str, use_angle_cls: bool
) -> str:
"""Return all detected text joined in reading order by newlines."""
text, _ = self.ocr_text_conf(
img_bgr, lang=lang, use_angle_cls=use_angle_cls
)
return text
def ocr_text_conf(
self, img_bgr: np.ndarray, *, lang: str, use_angle_cls: bool
):
"""Return ``(text, mean_confidence)`` for the image.
``mean_confidence`` is the average per-line recognition confidence in
``[0, 1]`` (or ``None`` if nothing was detected). It lets the UI flag
pages the OCR engine itself is unsure about.
"""
lines = self.ocr_lines(
img_bgr, lang=lang, use_angle_cls=use_angle_cls
)
text = "\n".join(line["text"] for line in lines)
if not lines:
return text, None
conf = sum(line["confidence"] for line in lines) / len(lines)
return text, float(conf)
# Module-level singleton.
engine = OCREngine()
|