Spaces:
Running
Running
Download normalizer.py from telecomsxchange/OpenTextShield: direct link, hf CLI and curl.
- Browser
- Download file 2.98 kB
-
https://huggingface.co/spaces/telecomsxchange/OpenTextShield/resolve/main/normalizer.py
- Command line
-
hf download hf://spaces/telecomsxchange/OpenTextShield/normalizer.py
-
curl -L -o normalizer.py https://huggingface.co/spaces/telecomsxchange/OpenTextShield/resolve/main/normalizer.py
2.98 kB
| """Adversarial text normalisation, matching the production OpenTextShield API. | |
| Trimmed standalone copy of EnhancedPreprocessor.normalize_unicode from | |
| https://github.com/TelecomsXChangeAPi/OpenTextShield | |
| (src/api_interface/services/enhanced_preprocessing.py), so the demo Space | |
| classifies obfuscated text the same way the deployed API does. | |
| """ | |
| import re | |
| import unicodedata | |
| # Zero-width / invisible formatting characters used in obfuscation attacks. | |
| INVISIBLE_CHARS = frozenset({ | |
| "", "", "", "", "", | |
| "", "", "", "", "", | |
| "", "", "", "͏", "", | |
| }) | |
| # Cyrillic and Greek letters that render like Latin ones. Folded only where | |
| # they are plausibly a spoof, never in real Russian, Ukrainian or Greek text. | |
| CONFUSABLES = { | |
| "а": "a", "с": "c", "ԁ": "d", "е": "e", "һ": "h", "і": "i", "ј": "j", | |
| "ӏ": "l", "о": "o", "р": "p", "ԛ": "q", "ѕ": "s", "ԝ": "w", "х": "x", | |
| "у": "y", | |
| "А": "A", "В": "B", "С": "C", "Е": "E", "Н": "H", "І": "I", "Ӏ": "I", | |
| "Ј": "J", "К": "K", "М": "M", "О": "O", "Р": "P", "Ԛ": "Q", "Ѕ": "S", | |
| "Т": "T", "Ԝ": "W", "Х": "X", "Ү": "Y", | |
| "α": "a", "ι": "i", "κ": "k", "ν": "v", "ο": "o", "ρ": "p", "τ": "t", | |
| "υ": "u", "χ": "x", | |
| "Α": "A", "Β": "B", "Ε": "E", "Ζ": "Z", "Η": "H", "Ι": "I", "Κ": "K", | |
| "Μ": "M", "Ν": "N", "Ο": "O", "Ρ": "P", "Τ": "T", "Υ": "Y", "Χ": "X", | |
| } | |
| def _script(ch: str) -> str: | |
| try: | |
| return unicodedata.name(ch).split(" ", 1)[0] | |
| except ValueError: | |
| return "" | |
| def _fold_compat_alnum(ch: str) -> str: | |
| if ch.isascii() or unicodedata.category(ch)[0] not in "LN": | |
| return ch | |
| folded = unicodedata.normalize("NFKC", ch) | |
| return folded if len(folded) == 1 and folded.isascii() and folded.isalnum() else ch | |
| def _fold_spoofed_word(word: str, mostly_latin: bool) -> str: | |
| foreign = [ch for ch in word if ch.isalpha() and _script(ch) in ("CYRILLIC", "GREEK")] | |
| if not foreign or any(ch not in CONFUSABLES for ch in foreign): | |
| return word | |
| has_latin = any(ch.isalpha() and _script(ch) == "LATIN" for ch in word) | |
| if has_latin or mostly_latin: | |
| return "".join(CONFUSABLES.get(ch, ch) for ch in word) | |
| return word | |
| def normalize_unicode(text: str) -> str: | |
| """Undo text obfuscation without damaging real non-Latin text.""" | |
| text = unicodedata.normalize("NFC", text) | |
| text = "".join(ch for ch in text if ch not in INVISIBLE_CHARS) | |
| text = "".join(_fold_compat_alnum(ch) for ch in text) | |
| letters = [ch for ch in text if ch.isalpha()] | |
| latin = sum(1 for ch in letters if _script(ch) == "LATIN") | |
| mostly_latin = bool(letters) and latin * 2 > len(letters) | |
| if mostly_latin: | |
| text = "".join( | |
| unicodedata.normalize("NFKC", ch) if "!" <= ch <= "~" or ch == " " else ch | |
| for ch in text | |
| ) | |
| return re.sub(r"\S+", lambda m: _fold_spoofed_word(m.group(), mostly_latin), text) | |