Spaces:
Running
Running
| """ | |
| doi_utils.py — deterministic DOI extraction and ISSN validation. | |
| No LLM. No guessing. If a DOI cannot be extracted, we say so and ask the user. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| from urllib.parse import unquote, urlparse | |
| # DOIs are 10.<registrant>/<suffix>. The suffix may contain almost anything, | |
| # so we trim trailing punctuation and known URL cruft afterwards. | |
| DOI_RE = re.compile(r"(10\.\d{4,9}/[^\s\"'<>&?#]+)", re.I) | |
| _TRAILING = ".,;:)]}>'\"-" | |
| _SUFFIX_JUNK = ( | |
| "/full", "/abstract", "/pdf", "/epub", "/meta", "/html", | |
| ".full", ".pdf", ".abstract", ".long", | |
| ) | |
| ISSN_RE = re.compile(r"^(\d{4})-?(\d{3}[\dXx])$") | |
| def extract_doi(text: str) -> str: | |
| """ | |
| Pull a DOI out of a URL, a citation, or raw text. | |
| Handles publisher URL shapes: | |
| https://www.tandfonline.com/doi/full/10.1080/19322909.2023.2221477 | |
| https://link.springer.com/article/10.1007/s11192-023-04812-4 | |
| https://doi.org/10.1016/j.jclepro.2023.136775 | |
| https://onlinelibrary.wiley.com/doi/10.1002/adma.202301234 | |
| 10.1109/TCOMM.2023.1234567 | |
| Returns "" when no DOI is present. | |
| """ | |
| if not text: | |
| return "" | |
| s = unquote(str(text).strip()) | |
| m = DOI_RE.search(s) | |
| if not m: | |
| return "" | |
| doi = m.group(1) | |
| # Strip publisher URL suffixes appended after the DOI. | |
| low = doi.lower() | |
| for junk in _SUFFIX_JUNK: | |
| if low.endswith(junk): | |
| doi = doi[: -len(junk)] | |
| low = doi.lower() | |
| doi = doi.rstrip(_TRAILING) | |
| return doi if "/" in doi and len(doi) > 7 else "" | |
| def looks_like_url(text: str) -> bool: | |
| try: | |
| p = urlparse((text or "").strip()) | |
| return p.scheme in ("http", "https") and bool(p.netloc) | |
| except ValueError: | |
| return False | |
| def normalise_issn(raw: str) -> str: | |
| """Return a hyphenated, checksum-valid ISSN, or '' if invalid.""" | |
| m = ISSN_RE.match((raw or "").strip()) | |
| if not m: | |
| return "" | |
| digits = (m.group(1) + m.group(2)).upper() | |
| if not issn_checksum_ok(digits): | |
| return "" | |
| return f"{digits[:4]}-{digits[4:]}" | |
| def issn_checksum_ok(eight: str) -> bool: | |
| """ISSN mod-11 check digit. Catches typos before we hit the network.""" | |
| e = eight.replace("-", "").upper() | |
| if len(e) != 8: | |
| return False | |
| try: | |
| total = sum(int(e[i]) * (8 - i) for i in range(7)) | |
| except ValueError: | |
| return False | |
| remainder = total % 11 | |
| check = 0 if remainder == 0 else 11 - remainder | |
| expected = "X" if check == 10 else str(check) | |
| return e[7] == expected | |
| def doi_url(doi: str) -> str: | |
| return f"https://doi.org/{doi}" if doi else "" | |