beacon / backend /astroparse_api /equations.py
kiyer's picture
fix: close pymupdf doc in extract_equations; guard short anchor snippets; use grid layout for equation tag
0c7af8a
Raw
History Blame Contribute Delete
4.95 kB
"""Detect and clip display equations from PDF pages.
Detection at line granularity using page.get_text("dict"). Each line is
equation-like when it has a right-aligned (N) tag OR high math-glyph density
with low prose fraction and short length. Adjacent equation lines/blocks are
merged vertically. Clips are saved as eqN.png in the figures directory.
"""
import re
from pathlib import Path
import pymupdf
from .parse import _prose_fraction, demarkdown
_MATH_CHARS = frozenset(
"αβγδεζηθικλμνξοπρστυφχψωΓΔΘΛΞΠΣΦΨΩ"
"∫∑∏√∝≈≡≠≤≥±×÷·∞∂∇∈∉⊂⊃⊆⊇∪∩→←↑↓⟨⟩"
)
_EQ_TAG = re.compile(r"\(\d+\)\s*$")
_MATH_DENSITY_THRESH = 0.18
_PROSE_THRESH = 0.3
_MAX_EQ_LINE_LEN = 280
_MERGE_Y_GAP = 4.0
_MARGIN = 4
def _math_glyph_density(text: str) -> float:
chars = [c for c in text if not c.isspace()]
if not chars:
return 0.0
math_count = sum(1 for c in chars if c in _MATH_CHARS or c in "=+-_^/\\|{}[]<>")
return math_count / len(chars)
def _is_eq_line(line: dict) -> bool:
text = "".join(s.get("c", "") for s in line.get("spans", []))
text = text.strip()
if not text:
return False
if _EQ_TAG.search(text):
return True
density = _math_glyph_density(text)
prose = _prose_fraction(text)
return density > _MATH_DENSITY_THRESH and prose < _PROSE_THRESH and len(text) < _MAX_EQ_LINE_LEN
def _merge_regions(bboxes: list[tuple]) -> list[tuple]:
if not bboxes:
return []
merged = [list(bboxes[0])]
for x0, y0, x1, y1 in bboxes[1:]:
last = merged[-1]
if y0 - last[3] <= _MERGE_Y_GAP:
last[0] = min(last[0], x0)
last[2] = max(last[2], x1)
last[3] = max(last[3], y1)
else:
merged.append([x0, y0, x1, y1])
return [tuple(r) for r in merged]
def extract_equations(pdf_bytes: bytes, figures_dir: Path) -> list[dict]:
"""Detect display equations in each page and clip them to PNG.
Returns list of dicts: { id, page, tag, anchorSnippet, hasImage }
"""
equations = []
eq_count = 0
with pymupdf.open(stream=pdf_bytes, filetype="pdf") as doc:
for page_num, page in enumerate(doc):
blocks = page.get_text("dict", flags=pymupdf.TEXT_PRESERVE_WHITESPACE)["blocks"]
prev_text = ""
for block in blocks:
if block.get("type") != 0:
continue
eq_lines: list[tuple] = []
tag = ""
for line in block.get("lines", []):
line_text = "".join(s.get("c", "") for s in line.get("spans", []))
if _is_eq_line(line):
bbox = line.get("bbox", (0, 0, 0, 0))
eq_lines.append(tuple(bbox))
m = _EQ_TAG.search(line_text)
if m and not tag:
tag = m.group(0).strip()
if not eq_lines:
block_text = " ".join(
"".join(s.get("c", "") for s in line.get("spans", []))
for line in block.get("lines", [])
).strip()
if block_text:
prev_text = block_text
continue
regions = _merge_regions(eq_lines)
for region in regions:
eq_count += 1
eq_id = f"eq{eq_count}"
x0, y0, x1, y1 = region
clip = pymupdf.Rect(
max(0, x0 - _MARGIN), max(0, y0 - _MARGIN),
x1 + _MARGIN, y1 + _MARGIN,
)
mat = pymupdf.Matrix(2, 2)
pix = page.get_pixmap(clip=clip, matrix=mat, alpha=False)
png_path = figures_dir / f"{eq_id}.png"
pix.save(str(png_path))
words = demarkdown(prev_text).split()
anchor = " ".join(words[-6:]) if words else ""
equations.append({
"id": eq_id,
"page": page_num + 1,
"tag": tag,
"anchorSnippet": anchor,
"hasImage": png_path.is_file(),
})
return equations
def anchor_equations(equations: list[dict], paragraphs: list) -> list[dict]:
"""Assign afterPara by matching anchorSnippet to paragraph text."""
result = []
for eq in equations:
snippet = eq.get("anchorSnippet", "").strip().lower()
after_para = ""
if snippet and len(snippet) >= 15:
for p in paragraphs:
if snippet in p.text.lower():
after_para = p.id
break
result.append({**eq, "afterPara": after_para})
return result