| """保護段:把必須 byte-exact 保留的片段換成 PUA 佔位符,最後再還原。 | |
| 由 V1 的 runtime.py 抽出成獨立模組,讓 V3 推論路徑不必匯入 V1 | |
| (V1 需要 yaml、sites.yaml、model.py,在 pip/HF 環境是多餘的相依)。 | |
| """ | |
| from __future__ import annotations | |
| import re as _re | |
| # URL 的字元集必須明列,不能用 \S+:中文後面沒有空白分隔, | |
| # \S+ 會把「…id=972,最后浏览日期…」整段吞進 copy buffer, | |
| # 那段中文就完全繞過後續處理、簡體原樣輸出(實測過的真實 bug)。 | |
| COPY_EXACT = _re.compile( | |
| r"https?://[A-Za-z0-9\-._~:/?#\[\]@!$&'()*+;=%]+" | |
| r"|[A-Za-z0-9][\w.+-]*@[\w-]+(?:\.[\w-]+)*\.[A-Za-z]{2,}" # Email | |
| r"|`[^`]+`" # inline code | |
| r"|```[\s\S]*?```" # code fence | |
| r"|\bv?\d+\.\d+(?:\.\d+)*(?:-[\w.]+)?\b" # 版本號 | |
| r"|\bgit@[\w.-]+:[\w./-]+" # git remote | |
| ) | |
| PUA_BASE = 0xE000 | |
| PUA_MAX = 0xF8FF | |
| def protect(text: str) -> tuple[str, list[str]]: | |
| saved: list[str] = [] | |
| def sub(m): | |
| if len(saved) >= PUA_MAX - PUA_BASE: | |
| return m.group(0) | |
| saved.append(m.group(0)) | |
| return chr(PUA_BASE + len(saved) - 1) | |
| return COPY_EXACT.sub(sub, text), saved | |
| def restore(text: str, saved: list[str]) -> str: | |
| if not saved: | |
| return text | |
| return "".join( | |
| saved[ord(c) - PUA_BASE] if PUA_BASE <= ord(c) < PUA_BASE + len(saved) | |
| else c for c in text) | |