Danny
Bilingual humanize-text-model (EN T5-small + ZH Chinese T5-small)
7cb8aac verified
Raw
History Blame Contribute Delete
7.18 kB
"""Shared rule engine for the humanize-text model.
Conservative bilingual editing rules derived from the reference projects:
- spaces/humanizer-lite (phrase replacements + protected spans)
- trust-fixes/humanizer.py (multi-method pipeline philosophy)
The rules generate the supervised (ai_text -> human_text) pairs used to
fine-tune the seq2seq model, and can also run as a standalone fallback.
"""
import re
# --------------------------------------------------------------------------
# Protected spans: code, URLs, paths, numbers, quoted strings.
# --------------------------------------------------------------------------
PROTECTED_PATTERN = re.compile(
r"(`[^`]+`|https?://[^\s\u4e00-\u9fff,。!?;:、]+|(?:[A-Za-z]:)?(?:[/\\][\w.\-]+)+|"
r"\b\d+(?:\.\d+)?(?:%|[A-Za-z]+)?\b|\"[^\"]*\"|'[^']*')"
)
CJK_RE = re.compile(r"[\u4e00-\u9fff]")
def protect(text: str, lang: str = "auto"):
"""Replace protected spans with language-aware placeholders:
EN ``PROTECTED_n``, ZH ``【保护n】`` (both tokenizer-friendly)."""
protected = []
zh = (
lang == "zh"
or (lang == "auto" and len(CJK_RE.findall(text)) > 0)
)
def replace(match):
token = f"【保护{len(protected)}】" if zh else f"PROTECTED_{len(protected)}"
protected.append(match.group(0))
return token
return PROTECTED_PATTERN.sub(replace, text), protected
def restore(text: str, protected):
"""Restore placeholders (either format, tolerant of inserted spaces)."""
for index, value in enumerate(protected):
text = re.sub(rf"PROTECTED_{index}\b", lambda m: value, text)
text = re.sub(
rf"【\s*保\s*护\s*{index}\s*】", lambda m: value, text
)
return text
# --------------------------------------------------------------------------
# Formulaic (AI) phrases, English and Chinese.
# --------------------------------------------------------------------------
EN_PHRASES = {
"in today's rapidly evolving world": "today",
"in today's fast-paced world": "now",
"it is important to note that": "",
"it is worth noting that": "",
"it can be seen that": "",
"serves as a testament to": "shows",
"a testament to": "proof of",
"delve into": "look at",
"leverage": "use",
"leveraging": "using",
"seamless": "smooth",
"robust": "reliable",
"game-changer": "big improvement",
"groundbreaking": "new",
"cutting-edge": "modern",
"state-of-the-art": "modern",
"plays a crucial role in": "matters for",
"plays a crucial role": "matters",
"in the realm of": "in",
"in conclusion": "",
"moreover": "also",
"furthermore": "also",
"additionally": "also",
"it is essential to": "we need to",
"comprehensive approach": "broad approach",
"foster": "support",
"fostering": "supporting",
"undoubtedly": "",
"no longer optional but essential": "necessary",
"revolutionize": "change",
"revolutionizing": "changing",
"holistic": "full",
}
ZH_PHRASES = {
"值得注意的是": "",
"在当今快速发展的时代": "现在",
"综上所述": "",
"总而言之": "",
"由此可见": "",
"赋能": "帮助",
"助力": "帮助",
"降本增效": "省钱提效",
"闭环": "完整流程",
"无缝": "顺畅",
"发挥着重要作用": "很有用",
"发挥着至关重要的作用": "很重要",
"数字化转型": "数字化",
"必然趋势": "趋势",
"积极拥抱新技术": "用新技术",
"不断提升": "提高",
"持续优化": "改进",
"实现": "做到",
"进一步": "",
}
EN_OPENERS = ["Moreover,", "Furthermore,", "Additionally,", "In conclusion,"]
EN_EXTRA = {
"very": "",
"really": "",
"extremely": "",
"highly": "",
}
CONTRACTIONS = {
"it is": "it's",
"it will": "it'll",
"we are": "we're",
"we will": "we'll",
"you are": "you're",
"you will": "you'll",
"do not": "don't",
"does not": "doesn't",
"cannot": "can't",
"I am": "I'm",
"I will": "I'll",
"that is": "that's",
"there is": "there's",
"will not": "won't",
"it is not": "it isn't",
}
ZH_PARTICLES_END = ["了", "吧", "呢"]
ZH_PARTICLES_START = ["其实", "说实话", "我总觉得"]
# --------------------------------------------------------------------------
# Core rule application (single pass, deterministic).
# --------------------------------------------------------------------------
def humanize_rules(text: str, lang: str = "auto") -> str:
"""Apply conservative editing rules to a draft. Deterministic."""
protected_text, protected = protect(text, lang)
if lang in ("auto", "en"):
low = protected_text.lower()
for phrase, replacement in EN_PHRASES.items():
protected_text = re.sub(re.escape(phrase), replacement, protected_text, flags=re.I)
# Only apply opener removals when they start a sentence boundary.
for opener in EN_OPENERS:
protected_text = re.sub(
re.escape(opener) + r"\s*", "", protected_text, flags=re.I
)
for word, replacement in EN_EXTRA.items():
protected_text = re.sub(r"\b" + re.escape(word) + r"\s+", "", protected_text, flags=re.I)
for phrase, replacement in CONTRACTIONS.items():
protected_text = re.sub(
r"\b" + re.escape(phrase) + r"\b", replacement, protected_text, flags=re.I
)
if lang in ("auto", "zh"):
for phrase, replacement in ZH_PHRASES.items():
protected_text = protected_text.replace(phrase, replacement)
protected_text = re.sub(r",{2,}", ",", protected_text)
protected_text = re.sub(r"[ \t]{2,}", " ", protected_text)
protected_text = re.sub(r"\s+([,.;:!?,。;:!?])", r"\1", protected_text)
protected_text = re.sub(r"[,;]\s*\.", ".", protected_text)
protected_text = re.sub(r"\.\s*[,;]", ".", protected_text)
protected_text = re.sub(r"[,,]\s*(?=[。!?.!?])", "", protected_text)
protected_text = re.sub(r"\.\s+([a-z])", lambda m: ". " + m.group(1).upper(), protected_text)
protected_text = re.sub(r"([.!?。!?])\s*\1+", r"\1", protected_text)
protected_text = re.sub(r"\n{3,}", "\n\n", protected_text).strip(" ,,")
return restore(protected_text, protected)
def zh_add_colloquial(text: str) -> str:
"""Add light colloquial markers (used for data augmentation only)."""
if re.search(r"[\u4e00-\u9fff]", text):
text = re.sub(r"[。!?]", lambda m: m.group(0) + "呢", text, count=1)
text = re.sub(r"^", "其实,", text, count=1)
text = text.replace("今天", "今天吧")
text = text.replace("我觉得", "我总觉得")
return text
def en_add_colloquial(text: str) -> str:
"""Add light colloquial markers (used for data augmentation only)."""
text = re.sub(r"(^|\. )(I|we|he|she|they)\b", r"\1Honestly, \2", text, count=1)
text = text.replace("The ", "The ", 1)
if " quite " not in text:
text = re.sub(r"(\w+)\.$", r"\1, honestly.", text, count=1)
return text