from __future__ import annotations import io import re from dataclasses import dataclass from pypdf import PdfReader MAX_PDF_PAGES = 3 TITLE_SEARCH_LINES = 12 MAX_ABSTRACT_CHARS = 2800 ABSTRACT_HEADER_RE = re.compile(r"^abstract\b[:\s-]*(.*)$", re.IGNORECASE) SECTION_HEADER_PATTERNS = ( re.compile(r"^\d+(\.\d+)*\s+[A-Z]"), re.compile(r"^[IVXLC]+\.\s+[A-Z]"), re.compile( r"^(introduction|background|methods?|materials and methods?|results?|discussion|" r"conclusion|conclusions|references|keywords|index terms?)\b", re.IGNORECASE, ), ) NOISE_PATTERNS = ( re.compile(r"^arXiv:\S+", re.IGNORECASE), re.compile(r"^page \d+", re.IGNORECASE), re.compile(r"^(published as|accepted at|to appear in)\b", re.IGNORECASE), re.compile(r"^\d+$"), ) AFFILIATION_HINTS = ( "university", "institute", "department", "laboratory", "school", "college", "centre", "center", "faculty", ) LIGATURES = { "\ufb00": "ff", "\ufb01": "fi", "\ufb02": "fl", "\ufb03": "ffi", "\ufb04": "ffl", } @dataclass class PdfArticleFields: title: str abstract: str pages_read: int def _normalize_text(text: str) -> str: normalized = text for source, target in LIGATURES.items(): normalized = normalized.replace(source, target) normalized = re.sub(r"\b([A-Z])\s+([A-Z]{2,}\b)", r"\1\2", normalized) return normalized def _clean_line(raw_line: str) -> str: normalized = _normalize_text(raw_line).replace("\x00", " ") normalized = re.sub(r"\s+", " ", normalized).strip() if not normalized: return "" for pattern in NOISE_PATTERNS: if pattern.match(normalized): return "" return normalized def _has_alpha_content(line: str, min_chars: int = 8) -> bool: return sum(char.isalpha() for char in line) >= min_chars def _looks_like_author_or_affiliation(line: str) -> bool: lowered = line.lower() if "@" in line or "http" in lowered: return True if any(hint in lowered for hint in AFFILIATION_HINTS): return True if re.search(r"\b[a-z0-9._%+-]+@[a-z0-9.-]+\.[a-z]{2,}\b", lowered): return True words = [re.sub(r"[^A-Za-z]", "", token) for token in line.split()] capitalized_words = sum( bool(word) and word[0].isupper() and (len(word) == 1 or word[1:].islower()) for word in words ) if line.count(",") >= 2 and capitalized_words >= 4: return True if len(line.split()) <= 6 and sum(token.istitle() for token in line.split()) >= 2: return True return False def _is_section_heading(line: str) -> bool: for pattern in SECTION_HEADER_PATTERNS: if pattern.match(line): return True return bool(len(line) < 80 and line.isupper() and any(char.isalpha() for char in line)) def _extract_text_from_pdf(pdf_bytes: bytes, max_pages: int) -> tuple[str, int]: reader = PdfReader(io.BytesIO(pdf_bytes)) page_texts = [] pages_read = min(len(reader.pages), max_pages) for page_index in range(pages_read): page_text = reader.pages[page_index].extract_text() or "" if page_text.strip(): page_texts.append(page_text) if not page_texts: raise ValueError( "Could not extract text from the uploaded PDF. Try another PDF or paste the fields manually." ) return "\n".join(page_texts), pages_read def _prepare_lines(raw_text: str) -> list[str]: return [line for line in (_clean_line(item) for item in raw_text.splitlines()) if line] def _find_abstract_header(lines: list[str]) -> tuple[int | None, str]: for index, line in enumerate(lines[:80]): match = ABSTRACT_HEADER_RE.match(line) if match: return index, match.group(1).strip() return None, "" def _extract_title(lines: list[str], abstract_index: int | None) -> str: search_limit = abstract_index if abstract_index is not None else min(len(lines), TITLE_SEARCH_LINES) search_limit = max(1, min(search_limit, TITLE_SEARCH_LINES)) title_start = None for index, line in enumerate(lines[:search_limit]): if len(line) > 220: continue if not _has_alpha_content(line): continue if ABSTRACT_HEADER_RE.match(line): continue title_start = index break if title_start is None: raise ValueError( "Could not detect the article title in the uploaded PDF. Try another PDF or edit the fields manually." ) title_parts = [lines[title_start]] for candidate in lines[title_start + 1 : min(search_limit, title_start + 3)]: if _looks_like_author_or_affiliation(candidate): break if not _has_alpha_content(candidate, min_chars=5): break current_title = " ".join(title_parts) if len(current_title) + 1 + len(candidate) > 220: break title_parts.append(candidate) return " ".join(title_parts) def _extract_abstract(lines: list[str], abstract_index: int | None, inline_text: str) -> str: if abstract_index is None: raise ValueError( "Could not detect the abstract section in the uploaded PDF. Try another PDF or paste the fields manually." ) abstract_parts = [inline_text] if inline_text else [] for line in lines[abstract_index + 1 :]: if abstract_parts and _is_section_heading(line): break if abstract_parts and ABSTRACT_HEADER_RE.match(line): continue abstract_parts.append(line) if len(" ".join(abstract_parts)) >= MAX_ABSTRACT_CHARS: break abstract = re.sub(r"\s+", " ", " ".join(part for part in abstract_parts if part)).strip() if len(abstract) < 120: raise ValueError( "The uploaded PDF did not yield a usable abstract. Try another PDF or paste the abstract manually." ) return abstract def extract_article_fields_from_pdf( pdf_bytes: bytes, *, max_pages: int = MAX_PDF_PAGES, ) -> PdfArticleFields: raw_text, pages_read = _extract_text_from_pdf(pdf_bytes, max_pages=max_pages) lines = _prepare_lines(raw_text) if not lines: raise ValueError( "The uploaded PDF does not contain readable text on the first pages. Try another PDF." ) abstract_index, inline_abstract = _find_abstract_header(lines) title = _extract_title(lines, abstract_index) abstract = _extract_abstract(lines, abstract_index, inline_abstract) return PdfArticleFields(title=title, abstract=abstract, pages_read=pages_read)