Voltline's picture
Release VimeML V2.1 step40000 FP32 and Core ML INT8 (GPL-2.0)
29f25be verified
Raw History Blame Contribute Delete
1.79 kB
"""Conservative text-block cleaning rules."""
import re
import unicodedata
URL_ONLY = re.compile(r"(?:[β†’β†—βžœ>]\s*)?https?://\S+")
PAIRED_DECORATION = re.compile(r"([β—‡β—†β– β–‘β˜…β˜†]{2,})(.+?)\1")
NAVIGATION_SUFFIX = re.compile(r"[γ€‚οΌοΌŸ!?]\s*Next\s*$")
DECORATION_CHARS = set(" -=_*#~γƒ»β—‡β—†β– β–‘β˜…β˜†β†’β†β†‘β†“/\\|─━")
def clean_block(original: str) -> dict:
changes = []
flags = []
text = unicodedata.normalize("NFC", original)
text = text.replace("\r\n", "\n").replace("\r", "\n")
text = text.lstrip("\ufeff")
# Replace controls with spaces to avoid joining unrelated words.
text = "".join(
" " if unicodedata.category(char) == "Cc"
and char not in "\n\t" else char
for char in text
).strip()
if text != original:
changes.append("basic_normalization")
action = "keep"
reason = "no_definite_noise"
if not text:
action, reason = "drop", "empty"
elif URL_ONLY.fullmatch(text):
action, reason = "drop", "standalone_url"
elif all(char in DECORATION_CHARS for char in text):
action, reason = "drop", "decoration_only"
else:
match = PAIRED_DECORATION.fullmatch(text)
if match:
text = match.group(2).strip()
changes.append("paired_decoration_removed")
if "\ufffd" in text:
flags.append("replacement_character")
if NAVIGATION_SUFFIX.search(text):
flags.append("possible_navigation_suffix")
if flags:
action, reason = "review", "suspected_noise"
return {
"text": text,
"action": action,
"reason": reason,
"changes": changes,
"flags": flags,
}