dosemate / medication_data.py
DoseMate
Deploy DoseMate to Hugging Face Spaces
a0c836f
Raw History Blame Contribute Delete
4.01 kB
"""Loader and helpers for medicines.json.
Used by the schedule builder and the DDI checker. Reads the SAME data file the
chat pipeline ingests, but uses the structured fields directly (no vector
search) because scheduling needs exact dosing / meal / interaction fields.
"""
import json
import os
from functools import lru_cache
from typing import List, Dict, Optional
from rapidfuzz import process, fuzz
_DATA_PATH = os.path.join(os.path.dirname(__file__), "medicines.json")
@lru_cache(maxsize=1)
def _load() -> List[dict]:
with open(_DATA_PATH, "r", encoding="utf-8") as f:
return json.load(f)
@lru_cache(maxsize=1)
def _index() -> Dict[str, dict]:
"""Map lowercased drug_name AND generic_name -> record."""
idx = {}
for rec in _load():
for key in (rec.get("drug_name"), rec.get("generic_name")):
if key:
idx[key.lower()] = rec
return idx
def list_drugs() -> List[dict]:
"""Catalog for the picker UI."""
return [
{
"drug_name": r.get("drug_name"),
"generic_name": r.get("generic_name"),
"drug_class": r.get("drug_class"),
}
for r in _load()
]
def resolve_drug(name: str) -> Optional[dict]:
"""Resolve free-text drug name to its record via exact then fuzzy match."""
if not name:
return None
key = name.strip().lower()
idx = _index()
if key in idx:
return idx[key]
match = process.extractOne(key, list(idx.keys()), scorer=fuzz.WRatio)
if match and match[1] >= 80:
return idx[match[0]]
return None
def _name_tokens(rec: dict) -> List[str]:
"""Lowercased name tokens used to detect a drug inside an interaction string."""
tokens = []
for key in ("drug_name", "generic_name"):
val = rec.get(key)
if val:
tokens.append(val.lower())
# also the first word (e.g. "aspirin" from "aspirin (acetylsalicylic acid)")
tokens.append(val.lower().split()[0])
return list(set(tokens))
def find_ddi_pairs(selected_names: List[str]) -> List[dict]:
"""Detect Drug-Drug Interactions among a set of selected drugs.
interacting_drug in the data is often a class/list string
(e.g. "NSAIDs (ibuprofen, naproxen, aspirin...)"), so we test whether any
selected drug's name tokens appear as a substring of that string.
Returns one warning per detected pair (deduplicated).
"""
records = []
for name in selected_names:
rec = resolve_drug(name)
if rec:
records.append(rec)
warnings = []
seen = set()
for rec in records:
a_name = rec.get("drug_name")
for interaction in rec.get("major_drug_interactions", []) or []:
interacting_str = (interaction.get("interacting_drug") or "").lower()
if not interacting_str:
continue
for other in records:
if other is rec:
continue
b_name = other.get("drug_name")
# does the OTHER drug appear inside this interaction string?
if any(tok and tok in interacting_str for tok in _name_tokens(other)):
pair_key = tuple(sorted([a_name, b_name]))
if pair_key in seen:
continue
seen.add(pair_key)
warnings.append({
"drug_a": a_name,
"drug_b": b_name,
"severity": interaction.get("severity"),
"description": interaction.get("description"),
"clinical_effect": interaction.get("clinical_effect"),
"recommended_gap_hours": interaction.get("recommended_gap_hours"),
})
return warnings
def get_missed_dose_instructions(name: str) -> Optional[str]:
rec = resolve_drug(name)
return rec.get("missed_dose_instructions") if rec else None