Spaces:
Runtime error
Runtime error
File size: 7,825 Bytes
a753e74 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 | """PDF text and structure extraction using PyMuPDF with font-aware parsing."""
from __future__ import annotations
import re
from pathlib import Path
from typing import Any
try:
import fitz # PyMuPDF
PYMUPDF_AVAILABLE = True
except ImportError:
PYMUPDF_AVAILABLE = False
def extract_pdf_text(pdf_path: Path) -> dict[str, Any]:
result: dict[str, Any] = {
"full_text": None,
"pages": [],
"page_count": 0,
"title_candidates": [],
"abstract": None,
"authors_raw": [],
"section_headings": [],
"references_raw": [],
"figure_captions": [],
"table_captions": [],
"equations": [],
"extraction_complete": False,
"extraction_notes": [],
}
if not PYMUPDF_AVAILABLE:
result["extraction_notes"].append("PyMuPDF not installed; PDF extraction skipped.")
return result
if not pdf_path.exists():
result["extraction_notes"].append(f"PDF not found: {pdf_path}")
return result
try:
doc = fitz.open(str(pdf_path))
result["page_count"] = len(doc)
pages_text = []
page_dicts = []
for page in doc:
pages_text.append(page.get_text("text"))
page_dicts.append(page.get_text("dict"))
doc.close()
result["pages"] = pages_text
result["full_text"] = "\n".join(pages_text)
# Font-aware title extraction from page 1 dict
if page_dicts:
result["title_candidates"] = _extract_title_by_fontsize(page_dicts[0])
# Fallback: plain-text heuristics for the rest
_parse_structure(result, page_dicts)
result["extraction_complete"] = True
except Exception as e:
result["extraction_notes"].append(f"Extraction error: {e}")
return result
def _extract_title_by_fontsize(page_dict: dict) -> list[str]:
"""Extract title candidates from page 1 by finding the largest-font text spans."""
spans: list[tuple[float, str]] = []
for block in page_dict.get("blocks", []):
if block.get("type") != 0: # text block
continue
for line in block.get("lines", []):
for span in line.get("spans", []):
text = span.get("text", "").strip()
size = span.get("size", 0)
if text and size > 8 and len(text) > 4:
spans.append((size, text))
if not spans:
return []
max_size = max(s for s, _ in spans)
# Title spans are within 90% of the maximum font size
threshold = max_size * 0.90
title_parts: list[str] = []
for size, text in spans:
if size >= threshold:
# Skip clearly non-title content (page numbers, headers/footers)
if re.fullmatch(r"[\d\s\-–—/|]+", text):
continue
title_parts.append(text)
elif title_parts:
# Stop collecting once font drops significantly after first title chunk
break
if title_parts:
combined = " ".join(title_parts)
return [combined] + title_parts[:2]
return []
def _parse_structure(result: dict, page_dicts: list[dict]) -> None:
"""Heuristically identify key structural elements from extracted text."""
full_text = result["full_text"] or ""
# Abstract: find text between "Abstract" and first section heading
abstract_match = re.search(
r"(?:abstract|Abstract)\s*[\n\r]+(.*?)(?:\n\s*\n|\n\s*(?:1[.\s]|introduction|Introduction|keywords|Keywords))",
full_text,
re.DOTALL | re.IGNORECASE,
)
if abstract_match:
result["abstract"] = abstract_match.group(1).strip()[:2000]
# Authors: lines between title and abstract on page 1 (heuristic)
if result["pages"]:
page1 = result["pages"][0]
result["authors_raw"] = _extract_authors_from_page1(page1, result["title_candidates"])
# Section headings using font-size approach first, then regex fallback
headings = _extract_headings_by_font(page_dicts)
if not headings:
heading_pattern = re.compile(
r"^(?:\d+(?:\.\d+)?\s+[A-Z][A-Za-z\s\-:]{3,60}|[A-Z][A-Z\s]{5,60})$",
re.MULTILINE,
)
headings = heading_pattern.findall(full_text)[:30]
result["section_headings"] = headings[:30]
# Figure/table captions
fig_pattern = re.compile(r"(?:Fig(?:ure)?\.?\s*\d+[.:\s]+[^\n]{10,200})", re.IGNORECASE)
result["figure_captions"] = fig_pattern.findall(full_text)[:20]
tab_pattern = re.compile(r"(?:Table\s+\d+[.:\s]+[^\n]{10,200})", re.IGNORECASE)
result["table_captions"] = tab_pattern.findall(full_text)[:20]
# References section
ref_match = re.search(
r"(?:\nReferences\n|\nBibliography\n)(.*?)$", full_text, re.DOTALL | re.IGNORECASE
)
if ref_match:
ref_text = ref_match.group(1)
refs = re.split(r"\n(?=\[\d+\]|\d+\.\s)", ref_text)
result["references_raw"] = [r.strip() for r in refs if len(r.strip()) > 20][:100]
def _extract_authors_from_page1(page1_text: str, title_candidates: list[str]) -> list[str]:
"""Heuristically extract author names from page 1 text."""
lines = [ln.strip() for ln in page1_text.splitlines() if ln.strip()]
# Find where title ends
skip_until = 0
if title_candidates:
for i, line in enumerate(lines):
if any(tc.lower()[:30] in line.lower() for tc in title_candidates[:1]):
skip_until = i + 1
break
candidate_lines = lines[skip_until : skip_until + 12]
authors: list[str] = []
for line in candidate_lines:
# Stop at abstract/keywords/section headings
if re.match(r"(?:abstract|keywords?|introduction|\d+[\.\s])", line, re.IGNORECASE):
break
# Author names: typically mixed case, may contain commas, "and", superscripts stripped
cleaned = re.sub(r"[∗†‡§¶,\d]+", "", line).strip()
if cleaned and 3 < len(cleaned) < 80 and not re.search(r"@|http|www|\.", cleaned):
authors.append(cleaned)
return authors[:10]
def _extract_headings_by_font(page_dicts: list[dict]) -> list[str]:
"""Extract headings by identifying medium-large font text that looks like section titles."""
all_sizes: list[float] = []
for pd in page_dicts:
for block in pd.get("blocks", []):
if block.get("type") != 0:
continue
for line in block.get("lines", []):
for span in line.get("spans", []):
size = span.get("size", 0)
if size > 0:
all_sizes.append(size)
if not all_sizes:
return []
body_size = sorted(all_sizes)[len(all_sizes) // 2] # median = body text size
heading_threshold = body_size * 1.1 # headings are > 10% larger than body
headings: list[str] = []
seen: set[str] = set()
for pd in page_dicts:
for block in pd.get("blocks", []):
if block.get("type") != 0:
continue
block_text_parts: list[str] = []
max_size_in_block = 0.0
for line in block.get("lines", []):
for span in line.get("spans", []):
size = span.get("size", 0)
text = span.get("text", "").strip()
if size > max_size_in_block:
max_size_in_block = size
if text:
block_text_parts.append(text)
if max_size_in_block >= heading_threshold:
combined = " ".join(block_text_parts).strip()
if 4 < len(combined) < 120 and combined not in seen:
seen.add(combined)
headings.append(combined)
return headings[:30]
|