themis / phase1 /scripts /pdf_sources.py
vg15o2's picture
Moonley backend (HF Space build)
1d9bd9b
Raw
History Blame Contribute Delete
9.68 kB
"""Verified source-PDF resolution for the judgment viewer.
The open SCR registry contains a small number of keys whose objects exist and are
labelled ``application/pdf`` but whose payload is actually an HTML error page.
Treating map membership as PDF availability therefore creates a false-positive
"Official PDF" tab.
This module keeps all duplicate source candidates, verifies the payload with a
bounded byte-range request, and caches only the verification result. The browser
can then load the verified public source directly, preserving byte-range support
without routing a large PDF through the CPU Space.
"""
from __future__ import annotations
from dataclasses import asdict, dataclass
import json
import os
import re
import threading
import time
from typing import Dict, Iterable, List, Optional
import requests
DEFAULT_PDF_BASE = "https://indian-supreme-court-judgments.s3.ap-south-1.amazonaws.com"
OFFICIAL_SCR_SEARCH = "https://scr.sci.gov.in/scrsearch/"
PDF_MAGIC = b"%PDF"
PROBE_BYTES = 1024
MIN_PDF_BYTES = 1024
@dataclass(frozen=True)
class PdfStatus:
status: str
url: Optional[str] = None
reason: Optional[str] = None
size: Optional[int] = None
provider: Optional[str] = None
source_key: Optional[str] = None
fallback_available: bool = False
@property
def verified(self) -> bool:
return self.status == "verified"
def public_dict(self) -> dict:
data = asdict(self)
data["verified"] = self.verified
data["source_name"] = (
"Supreme Court Reports open registry (AWS Open Data)"
if self.provider == "aws_open_data"
else "Bharat Courts public archive"
if self.provider == "bharat_courts"
else "Supreme Court source archive"
)
data["official_search_url"] = OFFICIAL_SCR_SEARCH
return data
def _identity_key(value: object) -> str:
"""Normalize a public citation/identity without conflating case titles."""
return re.sub(r"[^A-Z0-9]+", " ", str(value or "").upper()).strip()
def _total_size(response: requests.Response) -> Optional[int]:
content_range = response.headers.get("content-range", "")
if "/" in content_range:
try:
return int(content_range.rsplit("/", 1)[1])
except (TypeError, ValueError):
pass
try:
return int(response.headers.get("content-length", ""))
except (TypeError, ValueError):
return None
class PdfSourceResolver:
"""Resolve and verify mapped PDFs without downloading the whole document."""
def __init__(
self,
map_path: str,
base_url: str = DEFAULT_PDF_BASE,
*,
request_timeout: tuple = (5, 15),
verified_ttl: int = 24 * 60 * 60,
invalid_ttl: int = 6 * 60 * 60,
temporary_ttl: int = 60,
):
self.base_url = base_url.rstrip("/")
self.request_timeout = request_timeout
self.verified_ttl = verified_ttl
self.invalid_ttl = invalid_ttl
self.temporary_ttl = temporary_ttl
self.sources: Dict[str, List[str]] = {}
self.archive_candidates: Dict[str, List[dict]] = {}
self._cache: Dict[str, tuple] = {}
self._lock = threading.Lock()
self._load(map_path)
def _load(self, map_path: str) -> None:
if not os.path.exists(map_path):
return
with open(map_path, encoding="utf-8") as fh:
for line in fh:
try:
row = json.loads(line)
doc_id = _identity_key(row["doc_id"])
year = str(row["year"])
path = str(row["path"])
except (KeyError, TypeError, ValueError, json.JSONDecodeError):
continue
url = f"{self.base_url}/data/pdf/year={year}/english/{path}_EN.pdf"
candidates = self.sources.setdefault(doc_id, [])
if url not in candidates:
candidates.append(url)
archive = self.archive_candidates.setdefault(doc_id, [])
record = {"year": year, "path": path, "source_key": doc_id}
if record not in archive:
archive.append(record)
@property
def mapped_count(self) -> int:
return len(self.sources)
@staticmethod
def _keys(doc_id: str, aliases: Optional[Iterable[object]] = None) -> list[str]:
keys = []
for value in [doc_id, *(aliases or [])]:
key = _identity_key(value)
if key and key not in keys:
keys.append(key)
return keys
def mapped(self, doc_id: str, aliases: Optional[Iterable[object]] = None) -> bool:
return any(key in self.sources for key in self._keys(doc_id, aliases))
def _resolved_candidates(
self, doc_id: str, aliases: Optional[Iterable[object]] = None
) -> list[tuple[str, str]]:
resolved = []
for key in self._keys(doc_id, aliases):
for url in self.sources.get(key, []):
item = (url, key)
if item not in resolved:
resolved.append(item)
return resolved
def archive_candidate(
self, doc_id: str, aliases: Optional[Iterable[object]] = None
) -> Optional[dict]:
"""Return a trusted year/path for Bharat Courts' tar fallback."""
for key in self._keys(doc_id, aliases):
candidates = self.archive_candidates.get(key, [])
if candidates:
return dict(candidates[-1])
return None
def _cached(self, cache_key: str) -> Optional[PdfStatus]:
with self._lock:
item = self._cache.get(cache_key)
if not item:
return None
expires, status = item
if expires <= time.monotonic():
self._cache.pop(cache_key, None)
return None
return status
def _store(self, cache_key: str, status: PdfStatus) -> PdfStatus:
if status.status == "verified":
ttl = self.verified_ttl
elif status.status == "temporarily_unavailable":
ttl = self.temporary_ttl
else:
ttl = self.invalid_ttl
with self._lock:
self._cache[cache_key] = (time.monotonic() + ttl, status)
return status
def probe(
self,
doc_id: str,
*,
aliases: Optional[Iterable[object]] = None,
force: bool = False,
) -> PdfStatus:
keys = self._keys(doc_id, aliases)
cache_key = "|".join(keys)
if not force:
cached = self._cached(cache_key)
if cached:
return cached
candidates = self._resolved_candidates(doc_id, aliases)
if not candidates:
return self._store(cache_key, PdfStatus("not_mapped", reason="no_pdf_mapping"))
invalid_reasons: List[str] = []
temporary_reasons: List[str] = []
# The old dictionary loader used the last duplicate row. Try that first
# for continuity, but retain earlier candidates as fallbacks.
for url, source_key in reversed(candidates):
response = None
try:
response = requests.get(
url,
headers={"Range": f"bytes=0-{PROBE_BYTES - 1}"},
stream=True,
allow_redirects=True,
timeout=self.request_timeout,
)
status_code = response.status_code
if status_code not in (200, 206):
reason = f"http_{status_code}"
if status_code >= 500 or status_code in (408, 429):
temporary_reasons.append(reason)
else:
invalid_reasons.append(reason)
continue
prefix = response.raw.read(PROBE_BYTES, decode_content=True)
size = _total_size(response)
if size is not None and size < MIN_PDF_BYTES:
invalid_reasons.append(f"too_small_{size}")
continue
if PDF_MAGIC not in prefix[:PROBE_BYTES]:
invalid_reasons.append("payload_is_not_pdf")
continue
return self._store(
cache_key,
PdfStatus(
"verified",
url=url,
size=size,
provider="aws_open_data",
source_key=source_key,
fallback_available=True,
),
)
except requests.RequestException as exc:
temporary_reasons.append(type(exc).__name__)
finally:
if response is not None:
response.close()
# If any candidate could not be checked, fail transiently rather than
# making the stronger claim that every mapped source is invalid.
if temporary_reasons:
return self._store(
cache_key,
PdfStatus(
"temporarily_unavailable",
reason=";".join(dict.fromkeys(temporary_reasons)),
fallback_available=True,
),
)
return self._store(
cache_key,
PdfStatus(
"invalid_source",
reason=";".join(dict.fromkeys(invalid_reasons)) or "source_probe_failed",
fallback_available=True,
),
)