Twinity-1 / twlat /quotes.py
JacobLinCool's picture
Twinity-1: weights, compiled dictionary, inference code
6cc3500 verified
Raw
History Blame Contribute Delete
2.79 kB
"""引號正規化(繞過 zhtw-mcp 的 quote-pairing 缺陷)。
zhtw-mcp `convert` 的引號重指派會**刪除字元**(見
reports/upstream-issue-zhtw-mcp-quotes.md):
『稲亭物怪録』。 → 稲亭物怪録。 ← 正確的括號被刪
„Ich bin deutsche“。 → „Ich bin deutsche。← 德文引號被刪
實測這是 TWLAT 輸出中最大的單一錯誤來源(佔錯誤編輯 37.2%、
涉及 31/298 題)。因此本模組自行做**確定性**的引號轉換,
再把結果遮罩起來讓規則層碰不到。
規則(依《重訂標點符號手冊》):
最外層 「」,內層 『』,再內層回到 「」,以此類推。
只轉換與 CJK 相鄰的 CN 彎引號;英文縮寫(it's)、德文引號(„…“)不動。
"""
from __future__ import annotations
import regex
# 只認漢字/假名/諺文,**不含 CJK 標點**:否則 „Ich bin deutsche“。 的 “
# 會因為後面接了「。」而被誤判為中文語境。
CJK = regex.compile(r"[\p{Han}\p{Hiragana}\p{Katakana}\p{Hangul}]")
OPEN_CN = {"“": 0, "‘": 1} # “ ‘
CLOSE_CN = {"”": 0, "’": 1} # ” ’
PAIRS = [("「", "」"), ("『", "』")] # 「」 『』
ALL_QUOTES = "“”‘’「」『』"
def _cjk_near(text: str, i: int) -> bool:
"""該引號是否處於 CJK 語境(前後任一側 2 字元內有 CJK)。"""
for j in (i - 2, i - 1, i + 1, i + 2):
if 0 <= j < len(text) and CJK.match(text[j]):
return True
return False
def normalize(text: str) -> str:
"""把 CN 彎引號轉為臺灣規範引號,維持巢狀層級,且**不刪除任何字元**。"""
out = list(text)
depth = 0
for i, ch in enumerate(text):
if ch in OPEN_CN:
if not _cjk_near(text, i):
continue # 英文/德文語境 → 不動
out[i] = PAIRS[depth % 2][0]
depth += 1
elif ch in CLOSE_CN:
if not _cjk_near(text, i):
continue
depth = max(0, depth - 1)
out[i] = PAIRS[depth % 2][1]
return "".join(out)
def mask(text: str, base: int = 0xE800):
"""把所有引號字元換成 PUA 佔位符,讓規則層碰不到。回傳 (masked, saved)。"""
saved: list[str] = []
buf = []
for ch in text:
if ch in ALL_QUOTES and len(saved) < 0xF8FF - base:
buf.append(chr(base + len(saved)))
saved.append(ch)
else:
buf.append(ch)
return "".join(buf), saved
def unmask(text: str, saved: list[str], base: int = 0xE800) -> str:
if not saved:
return text
return "".join(saved[ord(c) - base] if base <= ord(c) < base + len(saved) else c
for c in text)