fda-drug-lookup / normalize.py
GPUburnout's picture
Upload normalize.py with huggingface_hub
4c40baa verified
Raw
History Blame Contribute Delete
2.01 kB
"""
Input normalization, run BEFORE matching. Defuses the unicode/homoglyph/whitespace
evasion class both reviewers flagged: full-width text, soft hyphens, zero-width
chars, Cyrillic/Greek look-alike letters, diacritics, trailing punctuation/quotes.
"""
import re
import unicodedata
# Cyrillic / Greek letters that look like Latin -> fold to Latin (homoglyph attack)
_CONFUSABLES = {
# Cyrillic
"а": "a", "е": "e", "о": "o", "р": "p", "с": "c",
"х": "x", "у": "y", "і": "i", "ѕ": "s", "к": "k",
"м": "m", "н": "h", "т": "t", "в": "b",
# Greek
"α": "a", "ο": "o", "ρ": "p", "ε": "e", "υ": "u",
"κ": "k", "ν": "v", "τ": "t", "χ": "x",
}
_ZERO_WIDTH = {"​", "‌", "‍", "", "­", "⁠", "‎", "‏"}
_STRIP_SYMBOLS = {"®", "™", "©", "℠", "°"}
def normalize_input(text: str) -> str:
if not text:
return ""
# 1. NFKC: fold full-width (TYLENOL -> TYLENOL), ligatures, etc.
t = unicodedata.normalize("NFKC", text)
# 2. map confusable scripts to Latin
t = "".join(_CONFUSABLES.get(ch, ch) for ch in t)
# 3. drop zero-width / soft-hyphen / control chars (incl. null byte); keep spaces
t = "".join(
ch for ch in t
if ch not in _ZERO_WIDTH and ch not in _STRIP_SYMBOLS
and (ch == " " or unicodedata.category(ch)[0] != "C")
)
# 4. strip diacritics (Depakote <- Dépakote)
t = "".join(c for c in unicodedata.normalize("NFKD", t) if not unicodedata.combining(c))
# 5. collapse whitespace, strip surrounding quotes + trailing punctuation
t = re.sub(r"\s+", " ", t).strip()
t = t.strip("'\"“”‘’`")
t = re.sub(r"[?!.,;:]+$", "", t).strip()
return t
if __name__ == "__main__":
tests = ["TYLENOL", "ti­zanidine",
"хanax", "аspirin", "lipitor®", "Dépakote",
" metformin ", "'lipitor'", "metformin?", "Omeprazole\x00"]
for t in tests:
print(f"{t!r:30s} -> {normalize_input(t)!r}")