File size: 2,975 Bytes
26ad266
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
"""Adversarial text normalisation, matching the production OpenTextShield API.

Trimmed standalone copy of EnhancedPreprocessor.normalize_unicode from
https://github.com/TelecomsXChangeAPi/OpenTextShield
(src/api_interface/services/enhanced_preprocessing.py), so the demo Space
classifies obfuscated text the same way the deployed API does.
"""

import re
import unicodedata

# Zero-width / invisible formatting characters used in obfuscation attacks.
INVISIBLE_CHARS = frozenset({
    "​", "‌", "‍", "‎", "‏",
    "⁠", "⁡", "⁢", "⁣", "⁤",
    "", "­", "᠎", "͏", "؜",
})

# Cyrillic and Greek letters that render like Latin ones. Folded only where
# they are plausibly a spoof, never in real Russian, Ukrainian or Greek text.
CONFUSABLES = {
    "а": "a", "с": "c", "ԁ": "d", "е": "e", "һ": "h", "і": "i", "ј": "j",
    "ӏ": "l", "о": "o", "р": "p", "ԛ": "q", "ѕ": "s", "ԝ": "w", "х": "x",
    "у": "y",
    "А": "A", "В": "B", "С": "C", "Е": "E", "Н": "H", "І": "I", "Ӏ": "I",
    "Ј": "J", "К": "K", "М": "M", "О": "O", "Р": "P", "Ԛ": "Q", "Ѕ": "S",
    "Т": "T", "Ԝ": "W", "Х": "X", "Ү": "Y",
    "α": "a", "ι": "i", "κ": "k", "ν": "v", "ο": "o", "ρ": "p", "τ": "t",
    "υ": "u", "χ": "x",
    "Α": "A", "Β": "B", "Ε": "E", "Ζ": "Z", "Η": "H", "Ι": "I", "Κ": "K",
    "Μ": "M", "Ν": "N", "Ο": "O", "Ρ": "P", "Τ": "T", "Υ": "Y", "Χ": "X",
}


def _script(ch: str) -> str:
    try:
        return unicodedata.name(ch).split(" ", 1)[0]
    except ValueError:
        return ""


def _fold_compat_alnum(ch: str) -> str:
    if ch.isascii() or unicodedata.category(ch)[0] not in "LN":
        return ch
    folded = unicodedata.normalize("NFKC", ch)
    return folded if len(folded) == 1 and folded.isascii() and folded.isalnum() else ch


def _fold_spoofed_word(word: str, mostly_latin: bool) -> str:
    foreign = [ch for ch in word if ch.isalpha() and _script(ch) in ("CYRILLIC", "GREEK")]
    if not foreign or any(ch not in CONFUSABLES for ch in foreign):
        return word
    has_latin = any(ch.isalpha() and _script(ch) == "LATIN" for ch in word)
    if has_latin or mostly_latin:
        return "".join(CONFUSABLES.get(ch, ch) for ch in word)
    return word


def normalize_unicode(text: str) -> str:
    """Undo text obfuscation without damaging real non-Latin text."""
    text = unicodedata.normalize("NFC", text)
    text = "".join(ch for ch in text if ch not in INVISIBLE_CHARS)
    text = "".join(_fold_compat_alnum(ch) for ch in text)

    letters = [ch for ch in text if ch.isalpha()]
    latin = sum(1 for ch in letters if _script(ch) == "LATIN")
    mostly_latin = bool(letters) and latin * 2 > len(letters)
    if mostly_latin:
        text = "".join(
            unicodedata.normalize("NFKC", ch) if "!" <= ch <= "~" or ch == " " else ch
            for ch in text
        )
    return re.sub(r"\S+", lambda m: _fold_spoofed_word(m.group(), mostly_latin), text)