medbillcodes-api / app /datasource.py
medbillcodes-deploy
Deploy cloud pilot API
5cceba0
Raw
History Blame Contribute Delete
5.52 kB
"""Resolve + fetch the real OHIP source files from ontario.ca.
The ministry embeds a date in each filename (e.g.
`moh-ohip-fee-schedule-master-text-2026-06-02.txt`) that changes on every
update, so we scrape the landing page to discover the *current* URLs rather
than hard-coding them. Downloads are content-hashed so the refresh job can
skip re-parsing/re-embedding when nothing changed.
Compliance note: these are PUBLIC reference documents (fee schedules), never
patient data. For a strictly air-gapped clinic, set OHIP_LOCAL_* paths and the
downloader is bypassed entirely.
"""
from __future__ import annotations
import hashlib
import logging
import re
from dataclasses import dataclass
from pathlib import Path
import httpx
from .config import settings
logger = logging.getLogger(__name__)
# Link to the fixed-width Physician Fee Schedule Master, "Text format".
_FSM_RE = re.compile(
r'href="([^"]*fee-schedule-master-text-[^"]*\.txt)"', re.IGNORECASE
)
# Link to the Physician Schedule of Benefits PDF (the big descriptive doc).
_SOB_RE = re.compile(
r'href="([^"]*moh-schedule-benefit-[^"]*\.pdf)"', re.IGNORECASE
)
# Trailing YYYY-MM-DD in the filename, used to pick the most recent file.
_DATE_RE = re.compile(r"(\d{4}-\d{2}-\d{2})")
def _latest(urls: list[str]) -> str | None:
"""Pick the URL whose embedded YYYY-MM-DD date is newest."""
if not urls:
return None
def key(u: str) -> str:
m = _DATE_RE.findall(u)
return m[-1] if m else "0000-00-00"
return max(urls, key=key)
@dataclass
class SourceFile:
path: Path
sha256: str
changed: bool
url: str | None = None
def _data_dir() -> Path:
d = Path(settings.ohip_data_dir)
d.mkdir(parents=True, exist_ok=True)
return d
def _sha256(data: bytes) -> str:
return hashlib.sha256(data).hexdigest()
def _absolutize(url: str) -> str:
if url.startswith("http"):
return url
if url.startswith("/"):
return f"https://www.ontario.ca{url}"
return f"https://www.ontario.ca/{url}"
def _discover_urls() -> tuple[str | None, str | None]:
"""Scrape the landing page for the current FSM + SoB URLs."""
fsm_url, sob_urls = _discover_all()
return fsm_url, (sob_urls[0] if sob_urls else None)
def _discover_all() -> tuple[str | None, list[str]]:
"""Return the latest FSM URL and ALL SoB PDF URLs (newest first)."""
with httpx.Client(timeout=30.0, follow_redirects=True) as client:
html = client.get(settings.ohip_source_page).text
fsm_url = _latest([_absolutize(u) for u in _FSM_RE.findall(html)])
def date_key(u: str) -> str:
m = _DATE_RE.findall(u)
return m[-1] if m else "0000-00-00"
sob_urls = sorted(
{_absolutize(u) for u in _SOB_RE.findall(html)}, key=date_key, reverse=True
)
logger.info("Discovered FSM=%s and %d SoB PDF(s)", fsm_url, len(sob_urls))
return fsm_url, sob_urls
def _write_if_changed(name: str, data: bytes, url: str | None) -> SourceFile:
dest = _data_dir() / name
digest = _sha256(data)
hash_file = _data_dir() / f"{name}.sha256"
previous = hash_file.read_text().strip() if hash_file.exists() else None
changed = digest != previous
if changed:
dest.write_bytes(data)
hash_file.write_text(digest)
logger.info("Updated %s (%d bytes, sha256=%s…)", name, len(data), digest[:12])
else:
logger.info("%s unchanged (sha256=%s…)", name, digest[:12])
return SourceFile(path=dest, sha256=digest, changed=changed, url=url)
def _download(url: str) -> bytes:
with httpx.Client(timeout=120.0, follow_redirects=True) as client:
resp = client.get(url)
resp.raise_for_status()
return resp.content
def fetch_fsm() -> SourceFile:
"""Return the current FSM text file (downloaded or local override)."""
if settings.ohip_local_fsm_path:
data = Path(settings.ohip_local_fsm_path).read_bytes()
return _write_if_changed("fsm.txt", data, url=None)
if not settings.allow_network_download:
raise RuntimeError(
"Network download disabled and no OHIP_LOCAL_FSM_PATH provided."
)
fsm_url, _ = _discover_urls()
if not fsm_url:
raise RuntimeError("Could not locate the FSM text URL on the ministry page.")
return _write_if_changed("fsm.txt", _download(fsm_url), url=fsm_url)
def fetch_sob_pdfs() -> list[SourceFile]:
"""Return ALL published Schedule of Benefits PDFs, newest first.
Description coverage varies between editions (a code listed with an inline
description in one year may appear only in a fee matrix the next), so we
parse every edition and merge — newest wins, older fills gaps. Fees always
come from the current FSM, so an older description text is still accurate
for identification/embedding purposes.
"""
if settings.ohip_local_sob_pdf_path:
data = Path(settings.ohip_local_sob_pdf_path).read_bytes()
return [_write_if_changed("sob.pdf", data, url=None)]
if not settings.allow_network_download:
logger.warning("Skipping SoB PDFs: network disabled and no local path set.")
return []
_, sob_urls = _discover_all()
if not sob_urls:
logger.warning("Could not locate any SoB PDF URLs; descriptions will be sparse.")
return []
files: list[SourceFile] = []
for idx, url in enumerate(sob_urls):
files.append(_write_if_changed(f"sob_{idx}.pdf", _download(url), url=url))
return files