Spaces:
Running
Running
File size: 2,975 Bytes
26ad266 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 | """Adversarial text normalisation, matching the production OpenTextShield API.
Trimmed standalone copy of EnhancedPreprocessor.normalize_unicode from
https://github.com/TelecomsXChangeAPi/OpenTextShield
(src/api_interface/services/enhanced_preprocessing.py), so the demo Space
classifies obfuscated text the same way the deployed API does.
"""
import re
import unicodedata
# Zero-width / invisible formatting characters used in obfuscation attacks.
INVISIBLE_CHARS = frozenset({
"", "", "", "", "",
"", "", "", "", "",
"", "", "", "͏", "",
})
# Cyrillic and Greek letters that render like Latin ones. Folded only where
# they are plausibly a spoof, never in real Russian, Ukrainian or Greek text.
CONFUSABLES = {
"а": "a", "с": "c", "ԁ": "d", "е": "e", "һ": "h", "і": "i", "ј": "j",
"ӏ": "l", "о": "o", "р": "p", "ԛ": "q", "ѕ": "s", "ԝ": "w", "х": "x",
"у": "y",
"А": "A", "В": "B", "С": "C", "Е": "E", "Н": "H", "І": "I", "Ӏ": "I",
"Ј": "J", "К": "K", "М": "M", "О": "O", "Р": "P", "Ԛ": "Q", "Ѕ": "S",
"Т": "T", "Ԝ": "W", "Х": "X", "Ү": "Y",
"α": "a", "ι": "i", "κ": "k", "ν": "v", "ο": "o", "ρ": "p", "τ": "t",
"υ": "u", "χ": "x",
"Α": "A", "Β": "B", "Ε": "E", "Ζ": "Z", "Η": "H", "Ι": "I", "Κ": "K",
"Μ": "M", "Ν": "N", "Ο": "O", "Ρ": "P", "Τ": "T", "Υ": "Y", "Χ": "X",
}
def _script(ch: str) -> str:
try:
return unicodedata.name(ch).split(" ", 1)[0]
except ValueError:
return ""
def _fold_compat_alnum(ch: str) -> str:
if ch.isascii() or unicodedata.category(ch)[0] not in "LN":
return ch
folded = unicodedata.normalize("NFKC", ch)
return folded if len(folded) == 1 and folded.isascii() and folded.isalnum() else ch
def _fold_spoofed_word(word: str, mostly_latin: bool) -> str:
foreign = [ch for ch in word if ch.isalpha() and _script(ch) in ("CYRILLIC", "GREEK")]
if not foreign or any(ch not in CONFUSABLES for ch in foreign):
return word
has_latin = any(ch.isalpha() and _script(ch) == "LATIN" for ch in word)
if has_latin or mostly_latin:
return "".join(CONFUSABLES.get(ch, ch) for ch in word)
return word
def normalize_unicode(text: str) -> str:
"""Undo text obfuscation without damaging real non-Latin text."""
text = unicodedata.normalize("NFC", text)
text = "".join(ch for ch in text if ch not in INVISIBLE_CHARS)
text = "".join(_fold_compat_alnum(ch) for ch in text)
letters = [ch for ch in text if ch.isalpha()]
latin = sum(1 for ch in letters if _script(ch) == "LATIN")
mostly_latin = bool(letters) and latin * 2 > len(letters)
if mostly_latin:
text = "".join(
unicodedata.normalize("NFKC", ch) if "!" <= ch <= "~" or ch == " " else ch
for ch in text
)
return re.sub(r"\S+", lambda m: _fold_spoofed_word(m.group(), mostly_latin), text)
|