Spaces:
Running
Running
| """Resolve + fetch the real OHIP source files from ontario.ca. | |
| The ministry embeds a date in each filename (e.g. | |
| `moh-ohip-fee-schedule-master-text-2026-06-02.txt`) that changes on every | |
| update, so we scrape the landing page to discover the *current* URLs rather | |
| than hard-coding them. Downloads are content-hashed so the refresh job can | |
| skip re-parsing/re-embedding when nothing changed. | |
| Compliance note: these are PUBLIC reference documents (fee schedules), never | |
| patient data. For a strictly air-gapped clinic, set OHIP_LOCAL_* paths and the | |
| downloader is bypassed entirely. | |
| """ | |
| from __future__ import annotations | |
| import hashlib | |
| import logging | |
| import re | |
| from dataclasses import dataclass | |
| from pathlib import Path | |
| import httpx | |
| from .config import settings | |
| logger = logging.getLogger(__name__) | |
| # Link to the fixed-width Physician Fee Schedule Master, "Text format". | |
| _FSM_RE = re.compile( | |
| r'href="([^"]*fee-schedule-master-text-[^"]*\.txt)"', re.IGNORECASE | |
| ) | |
| # Link to the Physician Schedule of Benefits PDF (the big descriptive doc). | |
| _SOB_RE = re.compile( | |
| r'href="([^"]*moh-schedule-benefit-[^"]*\.pdf)"', re.IGNORECASE | |
| ) | |
| # Trailing YYYY-MM-DD in the filename, used to pick the most recent file. | |
| _DATE_RE = re.compile(r"(\d{4}-\d{2}-\d{2})") | |
| def _latest(urls: list[str]) -> str | None: | |
| """Pick the URL whose embedded YYYY-MM-DD date is newest.""" | |
| if not urls: | |
| return None | |
| def key(u: str) -> str: | |
| m = _DATE_RE.findall(u) | |
| return m[-1] if m else "0000-00-00" | |
| return max(urls, key=key) | |
| class SourceFile: | |
| path: Path | |
| sha256: str | |
| changed: bool | |
| url: str | None = None | |
| def _data_dir() -> Path: | |
| d = Path(settings.ohip_data_dir) | |
| d.mkdir(parents=True, exist_ok=True) | |
| return d | |
| def _sha256(data: bytes) -> str: | |
| return hashlib.sha256(data).hexdigest() | |
| def _absolutize(url: str) -> str: | |
| if url.startswith("http"): | |
| return url | |
| if url.startswith("/"): | |
| return f"https://www.ontario.ca{url}" | |
| return f"https://www.ontario.ca/{url}" | |
| def _discover_urls() -> tuple[str | None, str | None]: | |
| """Scrape the landing page for the current FSM + SoB URLs.""" | |
| fsm_url, sob_urls = _discover_all() | |
| return fsm_url, (sob_urls[0] if sob_urls else None) | |
| def _discover_all() -> tuple[str | None, list[str]]: | |
| """Return the latest FSM URL and ALL SoB PDF URLs (newest first).""" | |
| with httpx.Client(timeout=30.0, follow_redirects=True) as client: | |
| html = client.get(settings.ohip_source_page).text | |
| fsm_url = _latest([_absolutize(u) for u in _FSM_RE.findall(html)]) | |
| def date_key(u: str) -> str: | |
| m = _DATE_RE.findall(u) | |
| return m[-1] if m else "0000-00-00" | |
| sob_urls = sorted( | |
| {_absolutize(u) for u in _SOB_RE.findall(html)}, key=date_key, reverse=True | |
| ) | |
| logger.info("Discovered FSM=%s and %d SoB PDF(s)", fsm_url, len(sob_urls)) | |
| return fsm_url, sob_urls | |
| def _write_if_changed(name: str, data: bytes, url: str | None) -> SourceFile: | |
| dest = _data_dir() / name | |
| digest = _sha256(data) | |
| hash_file = _data_dir() / f"{name}.sha256" | |
| previous = hash_file.read_text().strip() if hash_file.exists() else None | |
| changed = digest != previous | |
| if changed: | |
| dest.write_bytes(data) | |
| hash_file.write_text(digest) | |
| logger.info("Updated %s (%d bytes, sha256=%s…)", name, len(data), digest[:12]) | |
| else: | |
| logger.info("%s unchanged (sha256=%s…)", name, digest[:12]) | |
| return SourceFile(path=dest, sha256=digest, changed=changed, url=url) | |
| def _download(url: str) -> bytes: | |
| with httpx.Client(timeout=120.0, follow_redirects=True) as client: | |
| resp = client.get(url) | |
| resp.raise_for_status() | |
| return resp.content | |
| def fetch_fsm() -> SourceFile: | |
| """Return the current FSM text file (downloaded or local override).""" | |
| if settings.ohip_local_fsm_path: | |
| data = Path(settings.ohip_local_fsm_path).read_bytes() | |
| return _write_if_changed("fsm.txt", data, url=None) | |
| if not settings.allow_network_download: | |
| raise RuntimeError( | |
| "Network download disabled and no OHIP_LOCAL_FSM_PATH provided." | |
| ) | |
| fsm_url, _ = _discover_urls() | |
| if not fsm_url: | |
| raise RuntimeError("Could not locate the FSM text URL on the ministry page.") | |
| return _write_if_changed("fsm.txt", _download(fsm_url), url=fsm_url) | |
| def fetch_sob_pdfs() -> list[SourceFile]: | |
| """Return ALL published Schedule of Benefits PDFs, newest first. | |
| Description coverage varies between editions (a code listed with an inline | |
| description in one year may appear only in a fee matrix the next), so we | |
| parse every edition and merge — newest wins, older fills gaps. Fees always | |
| come from the current FSM, so an older description text is still accurate | |
| for identification/embedding purposes. | |
| """ | |
| if settings.ohip_local_sob_pdf_path: | |
| data = Path(settings.ohip_local_sob_pdf_path).read_bytes() | |
| return [_write_if_changed("sob.pdf", data, url=None)] | |
| if not settings.allow_network_download: | |
| logger.warning("Skipping SoB PDFs: network disabled and no local path set.") | |
| return [] | |
| _, sob_urls = _discover_all() | |
| if not sob_urls: | |
| logger.warning("Could not locate any SoB PDF URLs; descriptions will be sparse.") | |
| return [] | |
| files: list[SourceFile] = [] | |
| for idx, url in enumerate(sob_urls): | |
| files.append(_write_if_changed(f"sob_{idx}.pdf", _download(url), url=url)) | |
| return files | |