citation-checker / checker /verifier.py
aitziberluis's picture
Upload 8 files
4275a6e verified
Raw
History Blame Contribute Delete
6.93 kB
"""Verificación de referencias contra bases de datos académicas gratuitas.
Orden de comprobación por referencia:
1. Si trae DOI, se consulta directamente en Crossref (prueba definitiva).
2. Búsqueda bibliográfica en Crossref (revistas y actas de congresos).
3. Búsqueda por título en arXiv (preprints, muy común en IA/física).
Veredictos:
- VERIFICADA: encontrada con título y año compatibles.
- DUDOSA: se encontró algo parecido pero con discrepancias (año, título).
- NO ENCONTRADA: nada razonablemente parecido en las fuentes consultadas.
Ojo: no es prueba de que sea falsa (libros y textos no indexados no
aparecen), pero es una señal fuerte para revisarla a mano.
"""
from __future__ import annotations
import re
import unicodedata
import xml.etree.ElementTree as ET
from difflib import SequenceMatcher
import requests
CROSSREF_WORKS = "https://api.crossref.org/works"
ARXIV_API = "https://export.arxiv.org/api/query"
# Crossref pide identificarse para entrar en su "polite pool" (mejor servicio).
HEADERS = {"User-Agent": "citation-checker/1.0 (https://github.com/aitziberluis/citation-checker)"}
TITLE_MATCH_THRESHOLD = 0.75
YEAR_TOLERANCE = 1 # el año online y el impreso difieren a veces en 1
def _normalize(s: str) -> str:
s = unicodedata.normalize("NFKD", s.lower())
s = "".join(c for c in s if not unicodedata.combining(c))
return re.sub(r"[^a-z0-9 ]", " ", s).strip()
def _title_similarity(a: str, b: str) -> float:
na, nb = _normalize(a), _normalize(b)
if not na or not nb:
return 0.0
# Contención: útil cuando comparamos un título contra la referencia entera.
if na in nb or nb in na:
return 1.0
return SequenceMatcher(None, na, nb).ratio()
def _years_compatible(ref_year: int | None, found_year: int | None) -> bool:
if ref_year is None or found_year is None:
return True
return abs(ref_year - found_year) <= YEAR_TOLERANCE
def _crossref_item_to_match(item: dict) -> dict:
title = (item.get("title") or [""])[0]
year = None
for field in ("published-print", "published-online", "issued"):
parts = item.get(field, {}).get("date-parts", [[None]])
if parts and parts[0] and parts[0][0]:
year = parts[0][0]
break
return {
"title": title,
"year": year,
"doi": item.get("DOI"),
"url": f"https://doi.org/{item['DOI']}" if item.get("DOI") else None,
"source": "Crossref",
}
def check_doi(doi: str) -> dict | None:
"""Consulta directa de un DOI en Crossref. None si el DOI no existe."""
try:
r = requests.get(f"{CROSSREF_WORKS}/{doi}", headers=HEADERS, timeout=20)
except requests.RequestException:
return None
if r.status_code != 200:
return None
return _crossref_item_to_match(r.json()["message"])
def search_crossref(query: str) -> list[dict]:
try:
r = requests.get(
CROSSREF_WORKS,
params={"query.bibliographic": query[:500], "rows": 5},
headers=HEADERS,
timeout=20,
)
r.raise_for_status()
except requests.RequestException:
return []
items = r.json().get("message", {}).get("items", [])
return [_crossref_item_to_match(i) for i in items]
def search_arxiv(query: str) -> list[dict]:
try:
r = requests.get(
ARXIV_API,
params={"search_query": f'ti:"{query[:200]}"', "max_results": 3},
headers=HEADERS,
timeout=20,
)
r.raise_for_status()
root = ET.fromstring(r.text)
except (requests.RequestException, ET.ParseError):
return []
ns = {"atom": "http://www.w3.org/2005/Atom"}
matches = []
for entry in root.findall("atom:entry", ns):
title = (entry.findtext("atom:title", "", ns) or "").strip()
published = entry.findtext("atom:published", "", ns) or ""
link = entry.findtext("atom:id", "", ns) or None
matches.append(
{
"title": re.sub(r"\s+", " ", title),
"year": int(published[:4]) if published[:4].isdigit() else None,
"doi": None,
"url": link,
"source": "arXiv",
}
)
return matches
def verify(ref: dict) -> dict:
"""Verifica una referencia extraída y devuelve el veredicto.
Entrada: dict del extractor (title, authors, year, doi, raw).
Salida: {"verdict", "match", "note"} donde match es el mejor candidato.
"""
# 1. El DOI es la vía rápida y definitiva.
if ref.get("doi"):
match = check_doi(ref["doi"])
if match:
query_text = ref.get("title") or ref["raw"]
if _title_similarity(match["title"], query_text) >= TITLE_MATCH_THRESHOLD:
return {"verdict": "VERIFICADA", "match": match, "note": "DOI válido y título coincidente"}
return {
"verdict": "DUDOSA",
"match": match,
"note": "El DOI existe pero apunta a un trabajo con otro título — posible DOI inventado o mal copiado",
}
return {
"verdict": "DUDOSA",
"match": None,
"note": "La referencia trae un DOI que no existe en Crossref",
}
# 2. Búsqueda bibliográfica: Crossref con la cita entera (su búsqueda
# bibliográfica está pensada para eso) y arXiv por título si lo tenemos.
title = ref.get("title")
compare_against = title or ref["raw"]
candidates = search_crossref(ref["raw"])
if title:
candidates += search_arxiv(title)
scored = [
(c, _title_similarity(c["title"], compare_against)) for c in candidates
]
good = [(c, s) for c, s in scored if s >= TITLE_MATCH_THRESHOLD]
best, best_score = max(scored, key=lambda cs: cs[1], default=(None, 0.0))
if good:
# Entre las coincidencias buenas, preferir la de año compatible: las
# reimpresiones y reediciones comparten título pero no fecha.
compatible = [(c, s) for c, s in good if _years_compatible(ref.get("year"), c.get("year"))]
if compatible:
match = max(compatible, key=lambda cs: cs[1])[0]
return {"verdict": "VERIFICADA", "match": match, "note": f"Encontrada en {match['source']}"}
match = max(good, key=lambda cs: cs[1])[0]
return {
"verdict": "DUDOSA",
"match": match,
"note": f"Título encontrado en {match['source']} pero el año no cuadra "
f"(cita {ref.get('year')}, publicado {match.get('year')})",
}
return {
"verdict": "NO ENCONTRADA",
"match": best if best_score >= 0.5 else None,
"note": "Sin coincidencias razonables en Crossref ni arXiv — revísala a mano "
"(los libros y textos no indexados pueden no aparecer)",
}