"""理解层 ``UnderstandingLayer``(intent-understanding-layer 任务 2–7)。 底座**横切能力**:在固定流水线(``extract → compute → explain``)之前运行,使用 注入式 LLM 完成「文档画像 + 意图分类 + 信息提取 + 回填校验」,产出强结构化、可审计 的《分析任务单》(:class:`~kernel.task_sheet.AnalysisTaskSheet`)。 合规取向(与设计 / 需求一致): - **不进行任何分析数值计算**(需求 1.4 / Property 4):理解层只产出「来源可定位的 提取项 + 分类 + 缺失/歧义 + 澄清决策」,所有分析数值由后续 ``compute``(纯 Python) 产出。 - **无 LLM 即澄清,不退启发式硬算**(需求 6.1 / Property 6):无可用 LLM 时返回澄清 任务单(``reason_code="no_llm"``),绝不回退到有风险的启发式自动分析。 - **确定性防护优先于 LLM 判断**(需求 2.6 / 4.3 / Property 2、3):即便 LLM 误判, 保留时间 / RRT / 峰面积列也不得当作稳定性时间点;表头 / 标签单元格不得当作测定值; 绝不注入静默默认值(规格 0.5 / 默认 CQA / [24,36]),缺失即标记缺失。 本模块仅依赖标准库与本仓库 kernel 类型,LLM 通过注入适配器访问,便于无 LLM / 无 UI 的确定性单元测试。 """ from __future__ import annotations import json import logging import re from typing import Any, Optional, Protocol logger = logging.getLogger(__name__) # =========================================================================== # 任务 2:理解层 LLM 适配器(含无 LLM 检测) # =========================================================================== class UnderstandingLLM(Protocol): """理解层所需的最小 LLM 能力契约(便于注入 mock 做确定性测试)。""" def profile_and_extract(self, text: str, goal: str) -> Optional[dict]: """对原文 + 目标做一次结构化理解,返回 dict 或 ``None``(不可用 / 解析失败)。""" ... #: 理解层向底座 LLM 下达的系统提示(仅做结构化理解,绝不计算分析数值)。 _UNDERSTANDING_SYSTEM = ( "你是药学文档理解助手。你的唯一职责是【理解与提取】,绝不进行任何数值计算或外推。" "请阅读用户上传的文档文本与分析目标,输出严格的 JSON,描述:文档中有哪些表、" "每张表的类型、是否为稳定性时间序列;用户意图(描述性梳理 descriptive_summary / " "货架期外推 shelf_life_extrapolation / 药辅料相容性 compatibility / 不确定 unknown);以及可在原文中逐字定位的" "提取项(每项给出 source_ref 原文片段)。严禁臆造任何不在原文中的数值。" ) _UNDERSTANDING_USER_TMPL = ( "【文档文本】\n{text}\n\n【分析目标】\n{goal}\n\n" "【提取规则(务必遵守)】\n" "1. 逐属性拆分:每个测定值必须是**独立**的一项,一项只承载**单一**质量属性的" "单一数值。严禁把同一行的多个属性 / 多个数值塞进一个 value(如禁止 " "\"0.5 | 0.05 | 60\")。\n" "2. 每项必须给出 group:包含 strength(规格,如 20μg)、batch(批次,如有)、" "attribute(质量属性名,如 膜厚 / 总杂 / 含量)、table(该项所属表名,如 " "含量均匀度 / 有关物质检测结果,用于区分同名列并支撑专项判定)。\n" "3. 每项必须给出可在原文逐字定位的 source_ref(仅含该项自身的原文片段)。\n" "4. 规格限度单独成项(field 用 spec_limit),并在 group.attribute 标明其约束的属性。\n" "5. 严禁臆造原文中不存在的任何数值。\n\n" "【输出 JSON 结构】\n" "{{\n" ' "tables": [{{"table_type": "...", "title": "...", "source_document": "...", ' '"is_stability_time_series": false, "columns": [], "sample_rows": []}}],\n' ' "intent": "descriptive_summary|shelf_life_extrapolation|compatibility|unknown",\n' ' "items": [\n' ' {{"field": "膜厚", "value": "0.05", "source_ref": "膜厚 0.05", ' '"group": {{"strength": "20μg", "batch": "B1", "attribute": "膜厚", "table": "物理特性检测结果"}}}},\n' ' {{"field": "spec_limit", "value": "总杂≤2.0%", "source_ref": "限度 总杂≤2.0%", ' '"group": {{"attribute": "总杂", "table": "有关物质检测结果"}}}}\n' " ],\n" ' "spec_limits_present": true\n' "}}\n只输出 JSON,不要解释。" ) class _LLMServiceAdapter: """把底座 ``svc.llm``(``LLMService.complete``)适配为 :class:`UnderstandingLLM`。""" def __init__(self, llm: Any) -> None: self._llm = llm def profile_and_extract(self, text: str, goal: str) -> Optional[dict]: complete = getattr(self._llm, "complete", None) if not callable(complete): return None # 截断超长文本,避免 token 超限(理解阶段无需全文)。 max_chars = 8000 clipped = text[:max_chars] + ("\n…[截断]" if len(text) > max_chars else "") user = _UNDERSTANDING_USER_TMPL.format(text=clipped, goal=goal or "(未提供)") try: result = complete(_UNDERSTANDING_SYSTEM, user, temperature=0.0, scope="understanding") except Exception as exc: # noqa: BLE001 - LLM 调用异常视为不可用 logger.info("理解层 LLM 调用失败:%s", exc) return None if not getattr(result, "ok", False): return None return _safe_parse_json(getattr(result, "content", "") or "") def make_understanding_llm(svc: Any) -> Optional[UnderstandingLLM]: """构造理解层 LLM 适配器;无可用 LLM 时返回 ``None``(驱动澄清模式,需求 6.1)。 判定「可用」的最小条件:``svc.llm`` 存在且具备可调用的 ``complete``。真实可用性 (密钥 / 网络)在调用时体现——失败会被 :meth:`_LLMServiceAdapter.profile_and_extract` 转为 ``None``,上层据此进入澄清。 """ llm = getattr(svc, "llm", None) if llm is None: return None if not callable(getattr(llm, "complete", None)): return None return _LLMServiceAdapter(llm) def _safe_parse_json(content: str) -> Optional[dict]: """从 LLM 文本中安全解析 JSON 对象;失败返回 ``None``(不抛异常)。""" if not content: return None # 优先提取 ```json ... ``` 代码块。 m = re.search(r"```(?:json)?\s*([\s\S]*?)\s*```", content) candidate = m.group(1) if m else content.strip() if not candidate.startswith("{"): # 退一步:截取首个 { 到末个 } 的范围。 start = candidate.find("{") end = candidate.rfind("}") if start < 0 or end <= start: return None candidate = candidate[start : end + 1] try: parsed = json.loads(candidate) except (ValueError, TypeError): return None return parsed if isinstance(parsed, dict) else None # =========================================================================== # 任务 3:确定性防护(词典与归一助手) # =========================================================================== #: 保留时间 / RRT / 峰面积等**非稳定性时间**的列名词典(需求 2.6 / Property 3)。 _RETENTION_LEXICON = ( "保留时间", "retention", "rrt", "峰面积", "peak area", "peakarea", "理论塔板", "拖尾因子", "分离度", ) #: 表头 / 标签类单元格词典:这些是**列名 / 结构性文字**,不得当作测定值(需求 2.6)。 _HEADER_LABEL_LEXICON = ( "样品名称", "名称", "限度", "结论", "项目", "批号", "编号", "接受标准", "标准", "规格", "时间", "time", ) def is_retention_like_column(header: str) -> bool: """判断列名是否属于「保留时间 / RRT / 峰面积」等非稳定性时间语义(需求 2.6)。 命中则其数值**永不**作为稳定性时间点,即使 LLM 误标也由本防护兜底。 """ h = (header or "").strip().lower() if not h: return False # 单位 "min" 仅在与保留时间语境共现时才算(避免误伤),但显式 min 列也排除作时间点。 if h in ("min", "保留时间min", "rt", "rt(min)"): return True return any(kw in h for kw in _RETENTION_LEXICON) def is_header_or_label_cell(cell: str, known_headers: Optional[list[str]] = None) -> bool: """判断单元格是否为表头 / 标签类文字(不得当作测定值,需求 2.6)。 - 命中通用表头 / 标签词典;或 - 等于本表已知列名之一(``known_headers``)。 """ c = (cell or "").strip() if not c: return False cl = c.lower() if known_headers: norm_headers = {str(h).strip().lower() for h in known_headers} if cl in norm_headers: return True return any(kw == cl or kw in c for kw in _HEADER_LABEL_LEXICON if kw) def looks_like_stability_timepoints(values: list, headers: Optional[list[str]] = None) -> bool: """判断一组数值是否像合法的稳定性时间点序列(需求 2.6 防护)。 合法时间点的确定性最小特征: - 至少 2 个数值; - 非负、且能解析为数值; - 单调非降(稳定性考察时间点递增); - 不全部相等(保留时间常彼此接近且非递增)。 若列名命中保留时间词典,则**直接判否**(最强防护)。 """ if headers and any(is_retention_like_column(h) for h in headers): return False nums: list[float] = [] for v in values: try: nums.append(float(str(v).strip())) except (TypeError, ValueError): return False if len(nums) < 2: return False if any(n < 0 for n in nums): return False if len(set(nums)) == 1: return False # 单调非降。 if any(b < a for a, b in zip(nums, nums[1:])): return False return True def no_silent_defaults( extracted: dict, missing_items: list[str], ) -> dict: """移除静默默认值:把缺失的规格 / 主 CQA / 目标时间点标记为缺失,绝不填默认。 (需求 4.3 / Property 2)入参 ``extracted`` 为初步提取的字段映射;本函数原地补充 ``missing_items`` 并返回**清洗后的** dict(不含 0.5 / 总杂质 / [24,36] 之类默认)。 """ cleaned = dict(extracted or {}) # 规格限度:仅当确有来源时保留;否则标记缺失(不填 0.5)。 spec = cleaned.get("spec_limit") if spec in (None, "", []) or not cleaned.get("spec_limit_source"): cleaned.pop("spec_limit", None) if "规格限度" not in missing_items: missing_items.append("规格限度") # 主 CQA:缺失即标记(不填 "总杂质")。 if not cleaned.get("primary_cqa"): cleaned.pop("primary_cqa", None) if "主要质量指标" not in missing_items: missing_items.append("主要质量指标") # 目标时间点:缺失即标记(不填 [24, 36])。 tps = cleaned.get("target_timepoints") if not tps: cleaned.pop("target_timepoints", None) if "目标时间点" not in missing_items: missing_items.append("目标时间点") return cleaned __all__ = [ "UnderstandingLLM", "make_understanding_llm", "is_retention_like_column", "is_header_or_label_cell", "looks_like_stability_timepoints", "no_silent_defaults", "DocumentProfiler", "IntentClassifier", "InformationExtractor", "BackReferenceValidator", "UnderstandingLayer", "harvest_observations", "build_intent_slots", "SLOT_PRIMARY_CQA", "SLOT_SPEC_LIMIT", "SLOT_TARGET_TIMEPOINTS", "SLOT_TARGET_ATTRIBUTES", ] # =========================================================================== # 任务 4:DocumentProfiler(多文档聚合) # =========================================================================== from kernel.task_sheet import ( # noqa: E402 - 置于助手之后,避免顶部循环 CLARIFY_DATA_INTENT_MISMATCH, CLARIFY_INTENT_UNKNOWN, CLARIFY_NO_LLM, AnalysisTaskSheet, ClarificationDecision, DetectedTable, DocumentProfile, ExtractedItem, Intent, IntentSlot, SlotState, TableType, ) #: 表类型关键词 → TableType(用于无 LLM / 校正 LLM 输出的确定性兜底)。 _TABLE_TYPE_LEXICON: tuple[tuple[TableType, tuple[str, ...]], ...] = ( (TableType.RELATED_SUBSTANCES, ("有关物质", "杂质", "related substance", "impurit")), (TableType.CONTENT_UNIFORMITY, ("含量均匀度", "content uniformity", "uniformity")), (TableType.DISSOLUTION, ("溶出", "dissolution")), (TableType.ASSAY, ("含量", "assay", "potency", "效价")), (TableType.PHYSICAL_PROPERTIES, ("物理特性", "性状", "physical", "外观", "硬度", "脆碎")), ) def _canon_table_type(title: str, columns: Optional[list[str]] = None) -> TableType: """按标题 / 列名关键词推断表类型(确定性,不依赖 LLM)。""" hay = (title or "") + " " + " ".join(columns or []) hay = hay.lower() for ttype, kws in _TABLE_TYPE_LEXICON: if any(kw in hay for kw in kws): return ttype return TableType.UNKNOWN class DocumentProfiler: """文档画像器:识别表类型、是否稳定性时序,并跨文档聚合(需求 2)。""" def __init__(self, llm: Optional[UnderstandingLLM] = None) -> None: self._llm = llm def profile( self, text: str, file_names: Optional[list[str]] = None, *, goal: str = "", ) -> DocumentProfile: """产出 :class:`DocumentProfile`。 优先使用 LLM 的表清单(若注入且可用),再用确定性防护**强制**校正:保留时间 / 峰面积列绝不标为稳定性时序(需求 2.6 / Property 3)。无 LLM 时退回标题/列名 关键词的确定性画像(仍受同一防护约束)。 """ file_names = file_names or [] tables: list[DetectedTable] = [] llm_out = self._llm.profile_and_extract(text, goal) if self._llm else None if llm_out and isinstance(llm_out.get("tables"), list): for raw in llm_out["tables"]: tables.append(self._build_table_from_llm(raw, file_names)) else: tables.extend(self._heuristic_tables(text, file_names)) # 确定性防护:保留时间 / 峰面积主导的表,强制 is_stability_time_series=False。 for t in tables: if any(is_retention_like_column(c) for c in t.columns): t.is_stability_time_series = False has_stability = any(t.is_stability_time_series for t in tables) notes: list[str] = [] if not has_stability: notes.append("未检测到稳定性时间序列数据。") return DocumentProfile(tables=tables, has_stability_time_series=has_stability, notes=notes) def _build_table_from_llm(self, raw: dict, file_names: list[str]) -> DetectedTable: raw = raw or {} title = str(raw.get("title", "")) columns = [str(c) for c in (raw.get("columns") or [])] # 表类型:信任 LLM,但用关键词兜底校正未知。 ttype = TableType.UNKNOWN try: ttype = TableType(str(raw.get("table_type"))) except (ValueError, TypeError): ttype = _canon_table_type(title, columns) if ttype is TableType.UNKNOWN: ttype = _canon_table_type(title, columns) src = str(raw.get("source_document") or (file_names[0] if file_names else "")) is_ts = bool(raw.get("is_stability_time_series", False)) if ttype is TableType.STABILITY_TIME_SERIES: is_ts = True return DetectedTable( table_type=ttype, title=title, source_document=src, is_stability_time_series=is_ts, columns=columns, sample_rows=[[str(c) for c in row] for row in (raw.get("sample_rows") or [])], ) def _heuristic_tables(self, text: str, file_names: list[str]) -> list[DetectedTable]: """无 LLM 时的确定性画像:按「=== File: name ===」分段 + 标题关键词识别表类型。 这是一个**保守**的画像——它不试图解析数值(那是 compute 的事),只识别文本中 出现了哪些类型的表,以支撑意图/澄清判定。 """ tables: list[DetectedTable] = [] segments = _split_by_file(text, file_names) for src, body in segments: for title in _candidate_table_titles(body): ttype = _canon_table_type(title) if ttype is TableType.UNKNOWN: continue tables.append(DetectedTable( table_type=ttype, title=title.strip(), source_document=src, is_stability_time_series=False, )) return tables def _split_by_file(text: str, file_names: list[str]) -> list[tuple[str, str]]: """按 ``=== File: name ===`` 分隔符切分文本为 ``[(source, body)]``。""" if not text: return [] parts = re.split(r"=== File:\s*(.*?)\s*===", text) if len(parts) <= 1: default_src = file_names[0] if file_names else "" return [(default_src, text)] out: list[tuple[str, str]] = [] # re.split 结构:[前导, name1, body1, name2, body2, ...] i = 1 while i < len(parts) - 1: out.append((parts[i].strip(), parts[i + 1])) i += 2 return out def _candidate_table_titles(body: str) -> list[str]: """启发式抽取可能的表标题行(短行、含已知表类型关键词)。""" titles: list[str] = [] for line in (body or "").split("\n"): s = line.strip() if not s or len(s) > 40: continue if _canon_table_type(s) is not TableType.UNKNOWN: titles.append(s) return titles # =========================================================================== # 任务 5:IntentClassifier # =========================================================================== _DESCRIPTIVE_KW = ("梳理", "总结", "汇总", "规律", "概览", "概述", "整理", "describe", "summary", "summarize", "overview") _SHELF_LIFE_KW = ("货架期", "有效期", "外推", "预测", "shelf life", "shelf-life", "extrapolat", "多少个月", "多少月") _COMPATIBILITY_KW = ( "相容性", "配伍", "相互作用", "相容", "compatibility", "compatible", "辅料筛选", "处方筛选", "excipient compatibility", "drug-excipient", ) class IntentClassifier: """意图分类器:LLM 优先 + 确定性关键词兜底(需求 3)。""" def __init__(self, llm: Optional[UnderstandingLLM] = None) -> None: self._llm = llm def classify( self, goal: str, profile: DocumentProfile, *, llm_hint: Optional[str] = None, has_smiles: bool = False, has_excipient: bool = False, ) -> Intent: """返回 :class:`Intent`。 - 若提供 ``llm_hint``(来自一次已发生的 LLM 调用),优先采用其可识别值。 - **相容性强信号**:同时提供 SMILES 与辅料(相容性 Skill 的输入签名)→ 直接判定 为 COMPATIBILITY(这是确定性输入信号,优先于易混的关键词)。 - 否则用目标文本关键词判定;相容性 / 外推 / 描述关键词;都未命中则 UNKNOWN。 """ if llm_hint: try: hinted = Intent(str(llm_hint)) # LLM 明确给出非 UNKNOWN 意图时采用;UNKNOWN 不应压制下方确定性强信号。 if hinted is not Intent.UNKNOWN: return hinted except (ValueError, TypeError): pass # 相容性输入签名(SMILES + 辅料):最强确定性信号。 if has_smiles and has_excipient: return Intent.COMPATIBILITY g = (goal or "").lower() has_compat = any(kw.lower() in g for kw in _COMPATIBILITY_KW) has_desc = any(kw in g for kw in _DESCRIPTIVE_KW) has_shelf = any(kw in g for kw in _SHELF_LIFE_KW) # 相容性关键词优先于描述/外推(语义更专一)。 if has_compat and not has_shelf: return Intent.COMPATIBILITY if has_shelf and not has_desc: return Intent.SHELF_LIFE if has_desc and not has_shelf: return Intent.DESCRIPTIVE_SUMMARY # 仅有 SMILES(无辅料)且无明确外推/描述意图 → 也倾向相容性。 if has_smiles and not (has_shelf or has_desc): return Intent.COMPATIBILITY # 二者都命中或都未命中:交由澄清(需求 3.4)。 return Intent.UNKNOWN # =========================================================================== # 任务 6:InformationExtractor + BackReferenceValidator # =========================================================================== # 确定性观测采集:列名词典(用于判定一行是否为含属性列名的表头)。 _NUM_TOKEN_RE = re.compile(r"-?\d+(?:\.\d+)?") #: 删失/不等式型观测:前导比较符(半角/全角)+ 数值(+可选单位),如 >100、<0.05、≤0.1%。 #: 这类值是合法 QC 测定结果(耐折度>100、杂质=|<=?|>=?|<|>)\s*-?\d+(?:\.\d+)?\s*[%a-zA-Zμ·/℃°]*$" ) _STRENGTH_RE = re.compile(r"\d+\s*(?:μg|ug|mg|g)\b", re.IGNORECASE) _BATCH_RE = re.compile(r"[A-Za-z]{2,}-?\d{4,}[-\w]*") def _is_observation_value(s: str) -> bool: """单元格是否为可采集的观测值:纯数值,或删失/不等式型数值(如 >100、<0.05)。""" s = (s or "").strip() if not s: return False return bool(_NUM_TOKEN_RE.fullmatch(s) or _CENSORED_OBS_RE.match(s)) #: 统计/中间计算类列名(描述性梳理应排除,避免污染,需求 7.2)。 #: 注:「百分含量」是 assay 相对标示量结果列(关键 CQA,90~110% 限度的判定对象), #: 不属于中间统计列,故**不**列入此词典(修复 assay 结果列被误丢)。 _HELPER_COLUMN_LEXICON = ( "a+2.2s", "a+", "样品量", "含量均值", "相对偏差", "稀释倍数", "峰面积", "rsd", "标准偏差", "sd", "理论塔板", "拖尾", ) def _is_helper_column(col: str) -> bool: """是否为统计/中间计算列(非真实质量属性)。""" c = (col or "").strip().lower() if not c: return True if c in ("a", "s"): # 含量均匀度表的中间量 A / S return True return any(k in c for k in _HELPER_COLUMN_LEXICON) def _header_attr_count(row: list) -> int: """行中"像属性名"的非数值单元格数量(用于识别表头行)。""" cnt = 0 for c in row: if c is None: continue s = str(c).strip() if not s: continue if _NUM_TOKEN_RE.fullmatch(s): continue cnt += 1 return cnt def _row_strength(row: list) -> str: """从一行中提取规格 token(任一单元格命中 数字+μg/mg)。""" for c in row: if c is None: continue m = _STRENGTH_RE.search(str(c)) if m: return m.group(0).replace(" ", "") return "" def _row_batch(row: list) -> str: for c in row: if c is None: continue m = _BATCH_RE.search(str(c)) if m: return m.group(0) return "" def harvest_observations(profile: "DocumentProfile", tables: Optional[list] = None) -> list[ExtractedItem]: """从**结构化表格网格**确定性采集逐属性观测(修复列错位,需求 4 / 7.2)。 ``tables`` 为 :class:`services.file_service.TableGrid` 列表(保留空单元格=None)。 对每张表:定位属性表头行 → 其后数据行按**真实列索引**与表头属性配对 → 跳过保留时间 / 统计辅助列。空单元格被保留为 None,列位置不漂移(这是相对"拍平文本"的关键修复)。 数值直接来自文件,天然满足防捏造。无 ``tables`` 时返回空(上层据此走澄清 / 兜底)。 """ grids = tables or [] items: list[ExtractedItem] = [] observed_attrs: set[str] = set() attrs_with_spec: set[str] = set() for grid in grids: rows = getattr(grid, "rows", None) or (grid.get("rows") if isinstance(grid, dict) else None) or [] title = getattr(grid, "title", "") or (grid.get("title", "") if isinstance(grid, dict) else "") # 定位表头行:第一行"属性名"非数值单元格 >= 2 且本身几乎不含纯数值。 header_idx = -1 for i, row in enumerate(rows): if _header_attr_count(row) >= 2: numeric = sum(1 for c in row if c is not None and _NUM_TOKEN_RE.fullmatch(str(c).strip())) if numeric == 0: header_idx = i break if header_idx < 0: continue header = [(str(c).strip() if c is not None else "") for c in rows[header_idx]] for row in rows[header_idx + 1:]: if not any(c is not None for c in row): continue label_cell = next((str(c) for c in row if c is not None), "") # —— 限度行:首列含"限度/接受标准/标准/规格",按列对齐抓规格限度 —— if _is_limit_row(label_cell): for idx, cell in enumerate(row): if cell is None or idx >= len(header): continue s = str(cell).strip() if not s or s in ("/", "—", "-", "/"): continue col = header[idx] if not col or is_retention_like_column(col) or _is_helper_column(col): continue if is_header_or_label_cell(col) or not _looks_like_limit_value(s): continue attribute = _clean_header_attribute(col) if not attribute: continue items.append(ExtractedItem( field="spec_limit", value=s, source_ref=f"{label_cell} {col} {s}".strip(), located=True, group={"attribute": attribute, "table": title}, source="llm", )) attrs_with_spec.add(attribute) continue strength = _row_strength(row) batch = _row_batch(row) # 数据行需有 规格/批号 标识,否则可能是另一段表头或说明行。 if not (strength or batch): continue for idx, cell in enumerate(row): if cell is None or idx >= len(header): continue s = str(cell).strip() if not _is_observation_value(s): continue col = header[idx] if not col or is_retention_like_column(col) or _is_helper_column(col): continue if is_header_or_label_cell(col): continue attribute = _clean_header_attribute(col) if not attribute: continue observed_attrs.add(attribute) items.append(ExtractedItem( field=attribute, value=s, source_ref=f"{label_cell} {col} {s}".strip(), located=True, group={"strength": strength, "batch": batch, "attribute": attribute, "table": title}, source="llm", # 来自文件确定性采集,等同可信来源 )) # —— 保守的散文式限度解析(仅在属性名与操作符**字面相邻**时关联,否则放弃)—— items.extend(_harvest_prose_limits(grids, observed_attrs, attrs_with_spec)) return items def _attr_core(attribute: str) -> str: """属性核心名:去掉尾部 % / 单位 / 空白,用于在散文中做字面匹配。""" s = re.sub(r"[%%\s]+$", "", str(attribute or "")).strip() s = re.sub(r"(mg/g|mg|g|s|mpa·s|mpa|cm|mm)$", "", s, flags=re.IGNORECASE).strip() return s def _harvest_prose_limits(grids: list, observed_attrs: set, attrs_with_spec: set) -> list[ExtractedItem]: """从散文式限度文字中**保守**提取规格限度(需求 7.4,零误判取向)。 规则(严格、宁可放弃也不误关联): - 只在某**已观测到的属性名**于一个子句中**紧邻**比较符 / 区间出现时才关联; - 已有列对齐限度的属性不再覆盖; - 形如"含右美托咪啶应为90~110%"(用 API 名而非属性名"含量")→ 字面不含"含量" → **放弃**; - "L=15.0""忽略限"等无明确属性归属 → **放弃**。 """ out: list[ExtractedItem] = [] if not observed_attrs: return out # 属性核心名 → 原属性名(取最短核心,避免误配)。 cores = {a: _attr_core(a) for a in observed_attrs} seen: set[str] = set(attrs_with_spec) for grid in grids: rows = getattr(grid, "rows", None) or (grid.get("rows") if isinstance(grid, dict) else None) or [] title = getattr(grid, "title", "") or (grid.get("title", "") if isinstance(grid, dict) else "") for row in rows: for cell in row: if cell is None: continue s = str(cell).strip() if not _has_limit_operator(s): continue for clause in re.split(r"[;;。\n]", s): clause = clause.strip() if not clause or not _has_limit_operator(clause): continue for attr, core in cores.items(): if attr in seen or not core: continue if _attr_adjacent_to_limit(core, clause): out.append(ExtractedItem( field="spec_limit", value=clause, source_ref=clause, located=True, group={"attribute": attr, "table": title}, source="llm", )) seen.add(attr) # —— assay 含量范围措辞:「含<活性成分>应为 90.0~110.0%」—— # 这是中文药典 assay 验收的通用表述(限度以活性成分名而非属性名 # 表述)。仅当文档确有 assay 百分含量列、且范围落在标示量量级 # (上界≥80)时关联,避免误绑水分/杂质等小范围限度。 assay_attr = _assay_percent_attr(observed_attrs) if assay_attr and assay_attr not in seen: am = _ASSAY_RANGE_RE.search(clause) if am and float(am.group(2)) >= 80.0: out.append(ExtractedItem( field="spec_limit", value=f"{am.group(1)}~{am.group(2)}%", source_ref=clause, located=True, group={"attribute": assay_attr, "table": title}, source="llm", )) seen.add(assay_attr) return out #: assay 含量范围措辞:「含<活性成分>(应)为 ~%」(限度以 API 名表述)。 _ASSAY_RANGE_RE = re.compile( r"含[^,。;,;\s]{0,20}?(?:应为|应是|为)\s*(\d+(?:\.\d+)?)\s*[~~\--]\s*(\d+(?:\.\d+)?)\s*%?" ) def _assay_percent_attr(observed_attrs) -> str: """在已观测属性中找 assay 相对标示量百分比列(优先「百分含量」,其次含%的「含量」)。""" for a in observed_attrs: if "百分含量" in str(a): return a for a in observed_attrs: if "含量" in str(a) and "%" in str(a): return a return "" def _has_limit_operator(s: str) -> bool: return bool(re.search(r"[≤≥<><>]=?|~|~|不大于|不小于|nmt|nlt", str(s or ""), re.IGNORECASE)) def _attr_adjacent_to_limit(core: str, clause: str) -> bool: """属性核心名是否在子句中**紧邻**(其后 0–6 字符内)出现比较符/区间+数值。""" if core not in clause: return False pat = re.escape(core) + r".{0,6}?(?:[≤≥<><>]=?|~|~|不大于|不小于|nmt|nlt)\s*\d" return bool(re.search(pat, clause, re.IGNORECASE)) def _is_limit_row(label_cell: str) -> bool: """首列是否表明这是一条"限度 / 接受标准"行。""" c = (label_cell or "").strip().lower() return any(k in c for k in ("限度", "接受标准", "标准", "规格", "spec", "acceptance", "limit")) def _looks_like_limit_value(s: str) -> bool: """单元格是否是一个**简洁的**限度表述(用于列对齐限度,排除长句散文)。 仅接受:起始即比较符+数字(≤12.0 / <120s)、区间(2000~7000)、或纯数值。 长句 / 散文(如"含右美托咪啶应为90~110%""应符合…规定(L=15.0)")一律拒绝, 交由保守的散文匹配器按字面相邻关联(避免列对齐误绑)。 """ s = (s or "").strip() if not s or len(s) > 12: return False if re.match(r"^[≤≥<><>]=?\s*\d", s): return True if re.match(r"^\d+(?:\.\d+)?\s*[~~]\s*\d", s): return True if _NUM_TOKEN_RE.fullmatch(s): return True return False def _clean_header_attribute(col: str) -> str: """清洗列名为属性名:去单位括号 / 斜杠单位后缀。""" s = re.sub(r"[//].*$", "", str(col or "")).strip() s = re.sub(r"[\((].*?[\))]", "", s).strip() return s class InformationExtractor: """信息提取器:把 LLM 的 items 规整为 :class:`ExtractedItem` 列表(需求 4)。 本层**不**判定是否可定位(那是 :class:`BackReferenceValidator` 的职责);它只负责 结构化整形,并保留每项的 ``source_ref`` 原文片段。无 LLM 时返回空列表(上层据此 判定缺失 / 进入澄清,绝不臆造)。 """ def __init__(self, llm: Optional[UnderstandingLLM] = None) -> None: self._llm = llm def extract( self, text: str, profile: DocumentProfile, intent: Intent, *, llm_out: Optional[dict] = None, goal: str = "", tables: Optional[list] = None, ) -> list[ExtractedItem]: out = llm_out if out is None and self._llm is not None: out = self._llm.profile_and_extract(text, goal) items: list[ExtractedItem] = [] if out and isinstance(out.get("items"), list): for raw in out["items"]: if not isinstance(raw, dict): continue field_name = str(raw.get("field", "")).strip() if not field_name: continue value = str(raw.get("value", "")).strip() # 防护:表头 / 标签类文字不得作为测定值(需求 2.6)。 if field_name == "value" and is_header_or_label_cell(value): continue items.append(ExtractedItem( field=field_name, value=value, source_ref=str(raw.get("source_ref", "")).strip(), located=True, group=dict(raw.get("group") or {}), )) # 结构化网格优先(关键架构修正):当文件解析出**结构化表格网格**时,以确定性 # 网格采集为权威来源——它按真实列索引对齐、数值直取自单元格(最强防捏造), # 且统一覆盖删失值 / 限度行 / assay 措辞等。LLM 擅长「理解 / 意图 / 表分型」, # 不擅长「穷举逐格提取」,故有网格时不依赖 LLM 逐项枚举(修复 LLM 路径绕过 # 确定性修复、且 LLM 常漏列 / 漏单位的问题)。 if tables: harvested = harvest_observations(profile, tables) if harvested: return harvested # 无结构化网格(粘贴文本 / 不可解析文件)→ 回退 LLM 逐项提取(经回填校验防捏造)。 return items class BackReferenceValidator: """回填校验器:确认提取项的**数值确实存在于原文**,丢弃疑似捏造项(需求 4.1 / 4.2 / Property 1)。 设计要点(修复「引文过严误杀」):防捏造的本质是「**数值必须来自原文**」,而非「引文 必须逐字连续」。因此校验**以数值为中心**: 1. 用户手动项(``source=="user_edited"``)豁免(责任在用户)。 2. 提取项的数值 token 若出现在原文(归一空白后)→ 判定可定位、保留。 3. 非数值项(如文字型规格 / 结论)退回到「``source_ref`` 子串」判定,保持原有严格度。 4. 均不满足 → 视为疑似捏造,丢弃并记入 missing。 """ _NUM_RE = re.compile(r"-?\d+(?:\.\d+)?") @staticmethod def validate(items: list[ExtractedItem], text: str) -> tuple[list[ExtractedItem], list[str]]: """返回 ``(kept, missing_fields)``。""" haystack = text or "" norm_hay = re.sub(r"\s+", "", haystack) kept: list[ExtractedItem] = [] missing: list[str] = [] for it in items: # 1) 用户手动修改 / 新增项豁免(责任在用户),保留标记供审计区分。 if getattr(it, "source", "llm") == "user_edited": it.located = True kept.append(it) continue if BackReferenceValidator._is_grounded(it, haystack, norm_hay): it.located = True kept.append(it) else: if it.field not in missing: missing.append(it.field) return kept, missing @staticmethod def _is_grounded(it: ExtractedItem, haystack: str, norm_hay: str) -> bool: """判定单项是否「来自原文」:以数值为中心,文字项回退引文子串。""" value = (it.value or "").strip() nums = BackReferenceValidator._NUM_RE.findall(value) if nums: # 数值项:要求其**全部**数值 token 都能在原文找到(防止部分编造)。 norm_nums_ok = all( BackReferenceValidator._number_in_text(n, haystack) for n in nums ) if norm_nums_ok: return True # 数值未全部命中时,再给引文子串一次机会(容忍单位/格式差异)。 # 文字型项(无数值,如文字规格 / 结论)或数值项兜底:引文子串判定。 ref = re.sub(r"\s+", "", (it.source_ref or "").strip()) if ref and ref in norm_hay: return True # 文字型且无引文:用 value 自身做子串判定(避免漏掉合法文字项)。 if not nums: nv = re.sub(r"\s+", "", value) return bool(nv and nv in norm_hay) return False @staticmethod def _number_in_text(num: str, haystack: str) -> bool: """数值 token 是否出现在原文(允许尾随零差异,如 0.5 ↔ 0.50)。""" if not num: return False if num in haystack: return True # 容忍浮点格式差异:把原文所有数值规整后比对。 try: target = float(num) except ValueError: return False for m in BackReferenceValidator._NUM_RE.findall(haystack): try: if float(m) == target: return True except ValueError: continue return False # =========================================================================== # 意图保真:意图槽位派生(带来源态 / 置信度,禁止静默默认) # =========================================================================== #: 槽位名常量(对外契约,供澄清 / 路由 / 计算消费)。 SLOT_PRIMARY_CQA = "primary_cqa" SLOT_SPEC_LIMIT = "spec_limit" SLOT_TARGET_TIMEPOINTS = "target_timepoints" #: 描述性梳理意图维度:用户希望梳理的目标质量属性集合(需求 15)。 SLOT_TARGET_ATTRIBUTES = "target_attributes" #: 常见主 CQA 关键词(用户在指令中明示主指标时识别)。 _CQA_KEYWORDS = ( "总杂", "单杂", "有关物质", "杂质", "含量", "溶出", "水分", "效价", "assay", "impurit", "dissolution", "potency", "moisture", ) #: 目标时间点:匹配「24个月 / 36 月 / 24月」等中文月份表述。 _MONTHS_RE = re.compile(r"(\d+)\s*个?\s*月") def _confidence_for(state: "SlotState") -> float: """来源态 → 默认置信度(STATED 最高、INFERRED 居中、MISSING 为 0)。""" return {SlotState.STATED: 1.0, SlotState.INFERRED: 0.6, SlotState.MISSING: 0.0}[state] def build_intent_slots( goal: str, intent: Intent, extracted_items: Optional[list[ExtractedItem]] = None, ) -> list[IntentSlot]: """派生带保真元信息的意图槽位(需求 3 / 5;意图保真核心)。 原则(确保用户真实意图被采纳): - 用户在 ``goal`` 中**明示** → ``STATED``(置信 1.0,最高优先)。 - 仅能由提取项 / 文档**推断** → ``INFERRED``(置信居中,必须回显来源供改正)。 - 既无明示也无可靠推断(或多候选歧义)→ ``MISSING``(触发选择性澄清,**绝不填默认**)。 刻意**不产生任何默认值**(不填 0.5 / 总杂质 / [24,36])——缺失即 ``MISSING``, 交由澄清回到用户。当前为货架期外推(``SHELF_LIFE``)与描述性梳理 (``DESCRIPTIVE_SUMMARY``)派生关键槽位;其它意图返回空列表(不强加槽位, 保持向后兼容)。 """ if intent is Intent.DESCRIPTIVE_SUMMARY: return _build_descriptive_slots(goal, extracted_items or []) if intent is not Intent.SHELF_LIFE: return [] g = goal or "" items = extracted_items or [] slots: list[IntentSlot] = [] # —— 槽位 1:目标时间点(影响外推结论 → 缺失必问)—— months = _MONTHS_RE.findall(g) if months: values = sorted({int(m) for m in months}) slots.append(IntentSlot( name=SLOT_TARGET_TIMEPOINTS, value=values, state=SlotState.STATED, confidence=_confidence_for(SlotState.STATED), evidence=_first_match_span(_MONTHS_RE, g), affects_result=True, )) else: slots.append(IntentSlot( name=SLOT_TARGET_TIMEPOINTS, value=None, state=SlotState.MISSING, confidence=0.0, affects_result=True, )) # —— 槽位 2:主 CQA(影响外推对象 → 缺失 / 歧义必问)—— cqa_in_goal = next((k for k in _CQA_KEYWORDS if k.lower() in g.lower()), "") if cqa_in_goal: slots.append(IntentSlot( name=SLOT_PRIMARY_CQA, value=cqa_in_goal, state=SlotState.STATED, confidence=_confidence_for(SlotState.STATED), evidence=cqa_in_goal, affects_result=True, )) else: candidates = _distinct_attributes(items) if len(candidates) == 1: slots.append(IntentSlot( name=SLOT_PRIMARY_CQA, value=candidates[0], state=SlotState.INFERRED, confidence=_confidence_for(SlotState.INFERRED), evidence=f"提取自文档属性:{candidates[0]}", affects_result=True, )) else: # 0 个 → 无依据;≥2 个 → 歧义。两者都不臆断,标记缺失待澄清。 slots.append(IntentSlot( name=SLOT_PRIMARY_CQA, value=None, state=SlotState.MISSING, confidence=0.0, affects_result=True, )) # —— 槽位 3:规格限度(影响「合规」判定,但缺失可优雅降级为「不做合规判定」)—— # affects_result=False:依信息价值原则,规格缺失不强制追问(避免过度澄清), # 仅以来源态透明呈现;下游据 spec_provided 决定是否做合规结论。 spec_item = next((it for it in items if it.field == "spec_limit" and it.value), None) goal_spec = bool(re.search(r"[≤≥<><>]=?\s*\d|不大于|不小于|nmt|nlt", g, re.IGNORECASE)) if goal_spec: slots.append(IntentSlot( name=SLOT_SPEC_LIMIT, value=None, state=SlotState.STATED, confidence=_confidence_for(SlotState.STATED), evidence="指令中含限度表述", affects_result=False, )) elif spec_item is not None: slots.append(IntentSlot( name=SLOT_SPEC_LIMIT, value=spec_item.value, state=SlotState.INFERRED, confidence=_confidence_for(SlotState.INFERRED), evidence=spec_item.source_ref or spec_item.value, affects_result=False, )) else: slots.append(IntentSlot( name=SLOT_SPEC_LIMIT, value=None, state=SlotState.MISSING, confidence=0.0, affects_result=False, )) return slots def _build_descriptive_slots(goal: str, items: list[ExtractedItem]) -> list[IntentSlot]: """为描述性梳理(``DESCRIPTIVE_SUMMARY``)派生意图槽位(需求 15.1 / 15.3)。 唯一关键意图维度:``target_attributes``——用户希望梳理哪些质量属性。 - 指令**点名**质量属性(命中 ``_CQA_KEYWORDS``)→ ``STATED``,``value`` 为去重属性名列表。 - 未点名 → ``MISSING``(``value=None``),但 ``affects_result=False``:描述性梳理未指定时 默认梳理**全部**属性是良性、透明的默认(非篡改意图的静默默认),故不强制澄清,仅在确认 面板透明呈现供用户可选收窄。**绝不臆造具体属性默认值**。 """ g = goal or "" gl = g.lower() mentioned: list[str] = [] for kw in _CQA_KEYWORDS: if kw.lower() in gl and kw not in mentioned: mentioned.append(kw) if mentioned: return [IntentSlot( name=SLOT_TARGET_ATTRIBUTES, value=mentioned, state=SlotState.STATED, confidence=_confidence_for(SlotState.STATED), evidence="、".join(mentioned), affects_result=False, )] return [IntentSlot( name=SLOT_TARGET_ATTRIBUTES, value=None, state=SlotState.MISSING, confidence=0.0, affects_result=False, )] def _distinct_attributes(items: list["ExtractedItem"]) -> list[str]: """从提取项收集去重的质量属性名(排除规格限度项与空值)。""" seen: list[str] = [] for it in items: if it.field == "spec_limit": continue attr = str((it.group or {}).get("attribute") or "").strip() if attr and attr not in seen: seen.append(attr) return seen def _first_match_span(pattern: re.Pattern, text: str) -> str: """返回首个匹配的原文片段(供槽位 evidence 溯源),无匹配返回空串。""" m = pattern.search(text or "") return m.group(0) if m else "" # =========================================================================== # 任务 7:UnderstandingLayer.understand 编排 + 澄清门禁 # =========================================================================== #: 意图 → 建议 Skill id(与 Router.INTENT_TO_SKILL 保持一致)。 _INTENT_TO_SKILL = { Intent.DESCRIPTIVE_SUMMARY: "descriptive_summary", Intent.SHELF_LIFE: "stability", Intent.COMPATIBILITY: "compatibility", } class UnderstandingLayer: """理解层编排器:产出《分析任务单》或澄清任务单(需求 1 / 6)。""" def __init__(self, svc: Any) -> None: self.svc = svc def understand(self, raw: Any) -> AnalysisTaskSheet: """Phase-A 入口:纯编排、无分析数值计算(需求 1.4 / Property 4)。 流程:无 LLM → 澄清;汇集文本;文档画像 → 意图分类 → 提取 → 回填校验; 数据/意图不匹配 → 澄清;意图未知 → 澄清;任何异常 → 澄清(宁可多问)。 """ try: llm = make_understanding_llm(self.svc) if llm is None: return self._clarify(CLARIFY_NO_LLM, "understanding.clarify.no_llm") goal = str(getattr(raw, "goal", "") or "") file_names = [getattr(f, "name", "") or str(f) for f in (getattr(raw, "files", None) or [])] # 结构化摄入:保留表格网格(含空单元格),并拿到拍平文本供 LLM。 text, grids = self._gather_structured(raw) # 一次 LLM 调用,结果在画像 / 意图 / 提取间复用(确定性、省 token)。 llm_out = llm.profile_and_extract(text, goal) profiler = DocumentProfiler(llm) profile = self._profile_with_cached(profiler, text, file_names, goal, llm_out) # 确定性增强:转置 / 宽表稳定性数据(批次为列、时间点为行)LLM 常漏判。 # 用结构化网格做确定性识别,命中则强制标记为稳定性时序,避免「未检测到稳定性 # 时间序列数据」导致的货架期外推误澄清(修复转置宽表场景)。 self._augment_stability_from_grids(profile, grids, file_names) classifier = IntentClassifier(llm) intent = classifier.classify( goal, profile, llm_hint=(llm_out or {}).get("intent") if llm_out else None, has_smiles=bool(str(getattr(raw, "smiles", "") or "").strip()), has_excipient=bool(str(getattr(raw, "excipient", "") or "").strip()), ) extractor = InformationExtractor(llm) raw_items = extractor.extract( text, profile, intent, llm_out=llm_out, goal=goal, tables=grids ) kept, missing = BackReferenceValidator.validate(raw_items, text) # 意图未知 → 澄清(需求 3.4 / 6.4)。 if intent is Intent.UNKNOWN: sheet = self._clarify(CLARIFY_INTENT_UNKNOWN, "understanding.clarify.intent_unknown") sheet.document_profile = profile sheet.extracted_items = kept sheet.missing_items = missing self._audit("understanding_clarify", CLARIFY_INTENT_UNKNOWN) return sheet # 数据 / 意图不匹配:货架期外推但无稳定性时序 → 澄清(需求 6.3)。 if intent is Intent.SHELF_LIFE and not profile.has_stability_time_series: sheet = self._clarify( CLARIFY_DATA_INTENT_MISMATCH, "understanding.clarify.data_intent_mismatch" ) sheet.intent = intent sheet.document_profile = profile sheet.extracted_items = kept sheet.missing_items = missing self._audit("understanding_clarify", CLARIFY_DATA_INTENT_MISMATCH) return sheet sheet = AnalysisTaskSheet( intent=intent, document_profile=profile, extracted_items=kept, proposed_skill_id=_INTENT_TO_SKILL.get(intent, ""), missing_items=missing, source_method="llm", intent_slots=build_intent_slots(goal, intent, kept), ) self._audit("understanding_built", intent.value) return sheet except Exception as exc: # noqa: BLE001 - 任何异常转澄清,绝不硬算(需求 6) logger.warning("理解层处理异常,转入澄清:%s", exc, exc_info=True) return self._clarify(CLARIFY_INTENT_UNKNOWN, "understanding.clarify.error") # ------------------------------------------------------------------ @staticmethod def _augment_stability_from_grids( profile: DocumentProfile, grids: list, file_names: list[str] ) -> None: """用结构化网格确定性识别转置/宽表稳定性时序,命中则增强 profile(原地)。 转置宽表(批次为列、时间点为行,如「0天/加速1月/加速6月」)LLM 与启发式都常漏判。 本方法用 :func:`skills.stability.grid_parser.parse_stability_grids` 做确定性识别; 命中则设 ``has_stability_time_series=True``、补一张稳定性表、清除「未检测到」提示。 任何异常安全忽略(不改变原 profile)。 """ if profile.has_stability_time_series or not grids: return try: from skills.stability.grid_parser import parse_stability_grids data = parse_stability_grids(grids) except Exception: # noqa: BLE001 - 解析不可用时不增强 return if not data or not data.get("batches"): return src = file_names[0] if file_names else "" cqa = str(data.get("primary_cqa", "") or "稳定性") profile.tables.append(DetectedTable( table_type=TableType.STABILITY_TIME_SERIES, title=cqa, source_document=src, is_stability_time_series=True, columns=[cqa], )) profile.has_stability_time_series = True profile.notes = [n for n in (profile.notes or []) if "未检测到稳定性" not in n] # ------------------------------------------------------------------ @staticmethod def _profile_with_cached( profiler: "DocumentProfiler", text: str, file_names: list[str], goal: str, llm_out: Optional[dict], ) -> DocumentProfile: """复用已有的 llm_out 构建画像,避免二次 LLM 调用。""" tables: list[DetectedTable] = [] if llm_out and isinstance(llm_out.get("tables"), list): for raw_t in llm_out["tables"]: tables.append(profiler._build_table_from_llm(raw_t, file_names)) else: tables.extend(profiler._heuristic_tables(text, file_names)) for t in tables: if any(is_retention_like_column(c) for c in t.columns): t.is_stability_time_series = False has_stability = any(t.is_stability_time_series for t in tables) notes: list[str] = [] if has_stability else ["未检测到稳定性时间序列数据。"] return DocumentProfile(tables=tables, has_stability_time_series=has_stability, notes=notes) def _gather_text(self, raw: Any) -> str: """汇集上传文件文本(经 ``svc.file``)与直接粘贴文本。失败安全降级。""" text, _ = self._gather_structured(raw) return text def _gather_structured(self, raw: Any): """汇集为 ``(flattened_text, table_grids)``。 优先用 ``svc.file.parse_structured`` 拿到保留空单元格的表格网格(供确定性采集 按真实列索引配对);同时拼接拍平文本供 LLM。``parse_structured`` 不可用时退回 ``parse_text``(仅文本、无网格),保证向后兼容与稳健。 """ file_svc = getattr(self.svc, "file", None) parse_structured = getattr(file_svc, "parse_structured", None) parse_text = getattr(file_svc, "parse_text", None) parts: list[str] = [] grids: list = [] for f in (getattr(raw, "files", None) or []): name = getattr(f, "name", None) or str(f) try: path, cleanup = _materialize_file(f) try: if callable(parse_structured): doc = parse_structured(path) for g in getattr(doc, "tables", None) or []: grids.append(g) content = getattr(doc, "text", "") or "" elif callable(parse_text): content = parse_text(path) or "" else: content = "" finally: cleanup() except Exception as exc: # noqa: BLE001 logger.info("理解层结构化解析文件失败(%s):%s", name, exc) content = "" if content: parts.append(f"\n=== File: {name} ===\n{content}\n") pasted = (getattr(raw, "extra", None) or {}).get("text_content") if pasted: parts.append(str(pasted)) return "".join(parts), grids def _clarify(self, reason_code: str, message_key: str) -> AnalysisTaskSheet: return AnalysisTaskSheet( intent=Intent.UNKNOWN, clarification=ClarificationDecision( needed=True, reason_code=reason_code, message_key=message_key, options=["descriptive_summary", "shelf_life_extrapolation"], ), source_method="clarification", ) def _audit(self, event: str, detail: str) -> None: audit = getattr(self.svc, "audit", None) recorder = getattr(audit, "record_event", None) or getattr(audit, "record_analysis", None) if callable(recorder): try: recorder("understanding", event, detail) except Exception: # noqa: BLE001 - 审计失败不影响主流程 logger.debug("理解层审计记录失败(已忽略)。", exc_info=True) def _materialize_file(f): """把上传对象 / 路径解析为可被 FileService 读取的磁盘路径,返回 ``(path, cleanup)``。""" import os import tempfile def _noop() -> None: return None if isinstance(f, (str, os.PathLike)): p = str(f) if os.path.exists(p): return p, _noop name = getattr(f, "name", "") or "upload" suffix = os.path.splitext(name)[1] or "" data = None getvalue = getattr(f, "getvalue", None) if callable(getvalue): data = getvalue() else: read = getattr(f, "read", None) if callable(read): try: f.seek(0) except Exception: # noqa: BLE001 pass data = read() if data is None: return str(f), _noop if isinstance(data, str): data = data.encode("utf-8", errors="ignore") fd, tmp_path = tempfile.mkstemp(suffix=suffix) try: with os.fdopen(fd, "wb") as fh: fh.write(data) except Exception: # noqa: BLE001 return str(f), _noop def _cleanup() -> None: try: os.remove(tmp_path) except OSError: pass return tmp_path, _cleanup