"""Resolve + fetch the real OHIP source files from ontario.ca. The ministry embeds a date in each filename (e.g. `moh-ohip-fee-schedule-master-text-2026-06-02.txt`) that changes on every update, so we scrape the landing page to discover the *current* URLs rather than hard-coding them. Downloads are content-hashed so the refresh job can skip re-parsing/re-embedding when nothing changed. Compliance note: these are PUBLIC reference documents (fee schedules), never patient data. For a strictly air-gapped clinic, set OHIP_LOCAL_* paths and the downloader is bypassed entirely. """ from __future__ import annotations import hashlib import logging import re from dataclasses import dataclass from pathlib import Path import httpx from .config import settings logger = logging.getLogger(__name__) # Link to the fixed-width Physician Fee Schedule Master, "Text format". _FSM_RE = re.compile( r'href="([^"]*fee-schedule-master-text-[^"]*\.txt)"', re.IGNORECASE ) # Link to the Physician Schedule of Benefits PDF (the big descriptive doc). _SOB_RE = re.compile( r'href="([^"]*moh-schedule-benefit-[^"]*\.pdf)"', re.IGNORECASE ) # Trailing YYYY-MM-DD in the filename, used to pick the most recent file. _DATE_RE = re.compile(r"(\d{4}-\d{2}-\d{2})") def _latest(urls: list[str]) -> str | None: """Pick the URL whose embedded YYYY-MM-DD date is newest.""" if not urls: return None def key(u: str) -> str: m = _DATE_RE.findall(u) return m[-1] if m else "0000-00-00" return max(urls, key=key) @dataclass class SourceFile: path: Path sha256: str changed: bool url: str | None = None def _data_dir() -> Path: d = Path(settings.ohip_data_dir) d.mkdir(parents=True, exist_ok=True) return d def _sha256(data: bytes) -> str: return hashlib.sha256(data).hexdigest() def _absolutize(url: str) -> str: if url.startswith("http"): return url if url.startswith("/"): return f"https://www.ontario.ca{url}" return f"https://www.ontario.ca/{url}" def _discover_urls() -> tuple[str | None, str | None]: """Scrape the landing page for the current FSM + SoB URLs.""" fsm_url, sob_urls = _discover_all() return fsm_url, (sob_urls[0] if sob_urls else None) def _discover_all() -> tuple[str | None, list[str]]: """Return the latest FSM URL and ALL SoB PDF URLs (newest first).""" with httpx.Client(timeout=30.0, follow_redirects=True) as client: html = client.get(settings.ohip_source_page).text fsm_url = _latest([_absolutize(u) for u in _FSM_RE.findall(html)]) def date_key(u: str) -> str: m = _DATE_RE.findall(u) return m[-1] if m else "0000-00-00" sob_urls = sorted( {_absolutize(u) for u in _SOB_RE.findall(html)}, key=date_key, reverse=True ) logger.info("Discovered FSM=%s and %d SoB PDF(s)", fsm_url, len(sob_urls)) return fsm_url, sob_urls def _write_if_changed(name: str, data: bytes, url: str | None) -> SourceFile: dest = _data_dir() / name digest = _sha256(data) hash_file = _data_dir() / f"{name}.sha256" previous = hash_file.read_text().strip() if hash_file.exists() else None changed = digest != previous if changed: dest.write_bytes(data) hash_file.write_text(digest) logger.info("Updated %s (%d bytes, sha256=%s…)", name, len(data), digest[:12]) else: logger.info("%s unchanged (sha256=%s…)", name, digest[:12]) return SourceFile(path=dest, sha256=digest, changed=changed, url=url) def _download(url: str) -> bytes: with httpx.Client(timeout=120.0, follow_redirects=True) as client: resp = client.get(url) resp.raise_for_status() return resp.content def fetch_fsm() -> SourceFile: """Return the current FSM text file (downloaded or local override).""" if settings.ohip_local_fsm_path: data = Path(settings.ohip_local_fsm_path).read_bytes() return _write_if_changed("fsm.txt", data, url=None) if not settings.allow_network_download: raise RuntimeError( "Network download disabled and no OHIP_LOCAL_FSM_PATH provided." ) fsm_url, _ = _discover_urls() if not fsm_url: raise RuntimeError("Could not locate the FSM text URL on the ministry page.") return _write_if_changed("fsm.txt", _download(fsm_url), url=fsm_url) def fetch_sob_pdfs() -> list[SourceFile]: """Return ALL published Schedule of Benefits PDFs, newest first. Description coverage varies between editions (a code listed with an inline description in one year may appear only in a fee matrix the next), so we parse every edition and merge — newest wins, older fills gaps. Fees always come from the current FSM, so an older description text is still accurate for identification/embedding purposes. """ if settings.ohip_local_sob_pdf_path: data = Path(settings.ohip_local_sob_pdf_path).read_bytes() return [_write_if_changed("sob.pdf", data, url=None)] if not settings.allow_network_download: logger.warning("Skipping SoB PDFs: network disabled and no local path set.") return [] _, sob_urls = _discover_all() if not sob_urls: logger.warning("Could not locate any SoB PDF URLs; descriptions will be sparse.") return [] files: list[SourceFile] = [] for idx, url in enumerate(sob_urls): files.append(_write_if_changed(f"sob_{idx}.pdf", _download(url), url=url)) return files