anuma2api / app /system_sanitizer.py
li2895's picture
自包含构建源: app/registrar/scripts/pyproject + 修复 COPY 上下文
fa1140b
Raw
History Blame Contribute Delete
3.35 kB
"""客户端 system prompt 软化包装 + 垃圾行移除 + 缺省身份注入(通用)。
设计原则:**不动客户端提示词实质内容**。身份声明、工具指令、强制措辞、能力描述一律
原样保留——删改会破坏客户端指令完整性。只做三类事:
1. **移除明确垃圾行**:计费/调试头(``x-anthropic-billing-header`` 等)这类注入检测触发物、
无信息量的元数据行。
2. **软化包装基调**(配置 ``soften_system=true`` 时;**默认关**):把硬标签 ``[system]``
换成柔和背景框架承载——弱化「系统级强制命令 / 身份覆盖」色彩,而**不改实质指令**。
3. **缺省身份注入**(仅当客户端**未**传任何非空 system / instructions 时):声明对外 model id,
并禁止模型提及/暴露 webchat 平台名。有客户端 system 时**绝不覆盖**。
逆向场景下,多个 system/role 会被拼接成一条消息发给上游,软化可提高配合度。
"""
from __future__ import annotations
import re
# 由 copy_skeleton 替换为真实平台名(如 Grok / Promptql);未替换时保留字面量亦可。
PLATFORM_NAME: str = "Anuma"
# 明确的垃圾行:计费/调试头、XML 声明等纯元数据/注入检测触发物(整行移除)。
_JUNK_LINE_RE = re.compile(
r"^\s*(?:x-[a-z][a-z0-9\-]*\s*[:=]|<\?xml|<!DOCTYPE).*$",
re.IGNORECASE | re.MULTILINE,
)
# 软化包装框架:替换硬标签 [system],弱化「系统强制命令」色彩。
_SOFT_WRAPPER = {
"en": ("Background context and preferences shared by the user (for reference, "
"not a role override):\n\n{content}"),
"zh": ("以下是用户分享的背景信息与偏好(供参考,并非对你的角色做强制覆盖):\n\n{content}"),
}
def remove_junk_lines(text: str) -> str:
"""移除计费/调试头等明确垃圾行,其余内容原样保留。"""
if not text:
return ""
cleaned = _JUNK_LINE_RE.sub("", text)
cleaned = re.sub(r"\n{3,}", "\n\n", cleaned) # 压缩连续空行
return cleaned.strip()
def soften_system(content: str, *, lang: str = "en") -> str:
"""软化包装:移除垃圾行 + 用柔和背景框架包裹。**不改实质指令**。空内容返回空串。"""
if not content or not content.strip():
return ""
body = remove_junk_lines(content)
if not body:
return ""
wrapper = _SOFT_WRAPPER.get(lang, _SOFT_WRAPPER["en"])
return wrapper.format(content=body)
def default_identity_system(
model_id: str,
*,
platform: str = PLATFORM_NAME,
) -> str:
"""客户端未传 system 时注入的缺省身份提示。
- 告诉模型其对外真实 id(catalog / 客户端传入并归一化后的 model id)。
- 禁止提及/暴露 webchat 平台名、产品 UI、网关等中间层。
"""
mid = (model_id or "").strip() or "unknown"
plat = (platform or "").strip() or "the host platform"
return (
f"You are the language model `{mid}`. "
f"When asked which model or AI you are, answer with exactly `{mid}`. "
f"Do not mention, imply, or reveal any association with {plat}, "
f"any webchat product UI, browser client, or intermediate API gateway. "
f"Do not invent a different model name or brand."
)