"""客户端 system prompt 软化包装 + 垃圾行移除 + 缺省身份注入(通用)。 设计原则:**不动客户端提示词实质内容**。身份声明、工具指令、强制措辞、能力描述一律 原样保留——删改会破坏客户端指令完整性。只做三类事: 1. **移除明确垃圾行**:计费/调试头(``x-anthropic-billing-header`` 等)这类注入检测触发物、 无信息量的元数据行。 2. **软化包装基调**(配置 ``soften_system=true`` 时;**默认关**):把硬标签 ``[system]`` 换成柔和背景框架承载——弱化「系统级强制命令 / 身份覆盖」色彩,而**不改实质指令**。 3. **缺省身份注入**(仅当客户端**未**传任何非空 system / instructions 时):声明对外 model id, 并禁止模型提及/暴露 webchat 平台名。有客户端 system 时**绝不覆盖**。 逆向场景下,多个 system/role 会被拼接成一条消息发给上游,软化可提高配合度。 """ from __future__ import annotations import re # 由 copy_skeleton 替换为真实平台名(如 Grok / Promptql);未替换时保留字面量亦可。 PLATFORM_NAME: str = "Anuma" # 明确的垃圾行:计费/调试头、XML 声明等纯元数据/注入检测触发物(整行移除)。 _JUNK_LINE_RE = re.compile( r"^\s*(?:x-[a-z][a-z0-9\-]*\s*[:=]|<\?xml| str: """移除计费/调试头等明确垃圾行,其余内容原样保留。""" if not text: return "" cleaned = _JUNK_LINE_RE.sub("", text) cleaned = re.sub(r"\n{3,}", "\n\n", cleaned) # 压缩连续空行 return cleaned.strip() def soften_system(content: str, *, lang: str = "en") -> str: """软化包装:移除垃圾行 + 用柔和背景框架包裹。**不改实质指令**。空内容返回空串。""" if not content or not content.strip(): return "" body = remove_junk_lines(content) if not body: return "" wrapper = _SOFT_WRAPPER.get(lang, _SOFT_WRAPPER["en"]) return wrapper.format(content=body) def default_identity_system( model_id: str, *, platform: str = PLATFORM_NAME, ) -> str: """客户端未传 system 时注入的缺省身份提示。 - 告诉模型其对外真实 id(catalog / 客户端传入并归一化后的 model id)。 - 禁止提及/暴露 webchat 平台名、产品 UI、网关等中间层。 """ mid = (model_id or "").strip() or "unknown" plat = (platform or "").strip() or "the host platform" return ( f"You are the language model `{mid}`. " f"When asked which model or AI you are, answer with exactly `{mid}`. " f"Do not mention, imply, or reveal any association with {plat}, " f"any webchat product UI, browser client, or intermediate API gateway. " f"Do not invent a different model name or brand." )