File size: 1,791 Bytes
29f25be
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
"""Conservative text-block cleaning rules."""

import re
import unicodedata


URL_ONLY = re.compile(r"(?:[β†’β†—βžœ>]\s*)?https?://\S+")
PAIRED_DECORATION = re.compile(r"([β—‡β—†β– β–‘β˜…β˜†]{2,})(.+?)\1")
NAVIGATION_SUFFIX = re.compile(r"[γ€‚οΌοΌŸ!?]\s*Next\s*$")
DECORATION_CHARS = set(" -=_*#~γƒ»β—‡β—†β– β–‘β˜…β˜†β†’β†β†‘β†“/\\|─━")


def clean_block(original: str) -> dict:
    changes = []
    flags = []

    text = unicodedata.normalize("NFC", original)
    text = text.replace("\r\n", "\n").replace("\r", "\n")
    text = text.lstrip("\ufeff")

    # Replace controls with spaces to avoid joining unrelated words.
    text = "".join(
        " " if unicodedata.category(char) == "Cc"
        and char not in "\n\t" else char
        for char in text
    ).strip()

    if text != original:
        changes.append("basic_normalization")

    action = "keep"
    reason = "no_definite_noise"

    if not text:
        action, reason = "drop", "empty"
    elif URL_ONLY.fullmatch(text):
        action, reason = "drop", "standalone_url"
    elif all(char in DECORATION_CHARS for char in text):
        action, reason = "drop", "decoration_only"
    else:
        match = PAIRED_DECORATION.fullmatch(text)
        if match:
            text = match.group(2).strip()
            changes.append("paired_decoration_removed")

        if "\ufffd" in text:
            flags.append("replacement_character")
        if NAVIGATION_SUFFIX.search(text):
            flags.append("possible_navigation_suffix")

        if flags:
            action, reason = "review", "suspected_noise"

    return {
        "text": text,
        "action": action,
        "reason": reason,
        "changes": changes,
        "flags": flags,
    }