File size: 1,791 Bytes
29f25be | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 | """Conservative text-block cleaning rules."""
import re
import unicodedata
URL_ONLY = re.compile(r"(?:[βββ>]\s*)?https?://\S+")
PAIRED_DECORATION = re.compile(r"([βββ β‘β
β]{2,})(.+?)\1")
NAVIGATION_SUFFIX = re.compile(r"[γοΌοΌ!?]\s*Next\s*$")
DECORATION_CHARS = set(" -=_*#~γ»βββ β‘β
βββββ/\\|ββ")
def clean_block(original: str) -> dict:
changes = []
flags = []
text = unicodedata.normalize("NFC", original)
text = text.replace("\r\n", "\n").replace("\r", "\n")
text = text.lstrip("\ufeff")
# Replace controls with spaces to avoid joining unrelated words.
text = "".join(
" " if unicodedata.category(char) == "Cc"
and char not in "\n\t" else char
for char in text
).strip()
if text != original:
changes.append("basic_normalization")
action = "keep"
reason = "no_definite_noise"
if not text:
action, reason = "drop", "empty"
elif URL_ONLY.fullmatch(text):
action, reason = "drop", "standalone_url"
elif all(char in DECORATION_CHARS for char in text):
action, reason = "drop", "decoration_only"
else:
match = PAIRED_DECORATION.fullmatch(text)
if match:
text = match.group(2).strip()
changes.append("paired_decoration_removed")
if "\ufffd" in text:
flags.append("replacement_character")
if NAVIGATION_SUFFIX.search(text):
flags.append("possible_navigation_suffix")
if flags:
action, reason = "review", "suspected_noise"
return {
"text": text,
"action": action,
"reason": reason,
"changes": changes,
"flags": flags,
} |