File size: 59,995 Bytes
0e6887b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
"""理解层 ``UnderstandingLayer``(intent-understanding-layer 任务 2–7)。

底座**横切能力**:在固定流水线(``extract → compute → explain``)之前运行,使用
注入式 LLM 完成「文档画像 + 意图分类 + 信息提取 + 回填校验」,产出强结构化、可审计
的《分析任务单》(:class:`~kernel.task_sheet.AnalysisTaskSheet`)。

合规取向(与设计 / 需求一致):

- **不进行任何分析数值计算**(需求 1.4 / Property 4):理解层只产出「来源可定位的
  提取项 + 分类 + 缺失/歧义 + 澄清决策」,所有分析数值由后续 ``compute``(纯 Python)
  产出。
- **无 LLM 即澄清,不退启发式硬算**(需求 6.1 / Property 6):无可用 LLM 时返回澄清
  任务单(``reason_code="no_llm"``),绝不回退到有风险的启发式自动分析。
- **确定性防护优先于 LLM 判断**(需求 2.6 / 4.3 / Property 2、3):即便 LLM 误判,
  保留时间 / RRT / 峰面积列也不得当作稳定性时间点;表头 / 标签单元格不得当作测定值;
  绝不注入静默默认值(规格 0.5 / 默认 CQA / [24,36]),缺失即标记缺失。

本模块仅依赖标准库与本仓库 kernel 类型,LLM 通过注入适配器访问,便于无 LLM / 无 UI
的确定性单元测试。
"""

from __future__ import annotations

import json
import logging
import re
from typing import Any, Optional, Protocol

logger = logging.getLogger(__name__)


# ===========================================================================
# 任务 2:理解层 LLM 适配器(含无 LLM 检测)
# ===========================================================================

class UnderstandingLLM(Protocol):
    """理解层所需的最小 LLM 能力契约(便于注入 mock 做确定性测试)。"""

    def profile_and_extract(self, text: str, goal: str) -> Optional[dict]:
        """对原文 + 目标做一次结构化理解,返回 dict 或 ``None``(不可用 / 解析失败)。"""
        ...


#: 理解层向底座 LLM 下达的系统提示(仅做结构化理解,绝不计算分析数值)。
_UNDERSTANDING_SYSTEM = (
    "你是药学文档理解助手。你的唯一职责是【理解与提取】,绝不进行任何数值计算或外推。"
    "请阅读用户上传的文档文本与分析目标,输出严格的 JSON,描述:文档中有哪些表、"
    "每张表的类型、是否为稳定性时间序列;用户意图(描述性梳理 descriptive_summary / "
    "货架期外推 shelf_life_extrapolation / 药辅料相容性 compatibility / 不确定 unknown);以及可在原文中逐字定位的"
    "提取项(每项给出 source_ref 原文片段)。严禁臆造任何不在原文中的数值。"
)

_UNDERSTANDING_USER_TMPL = (
    "【文档文本】\n{text}\n\n【分析目标】\n{goal}\n\n"
    "【提取规则(务必遵守)】\n"
    "1. 逐属性拆分:每个测定值必须是**独立**的一项,一项只承载**单一**质量属性的"
    "单一数值。严禁把同一行的多个属性 / 多个数值塞进一个 value(如禁止 "
    "\"0.5 | 0.05 | 60\")。\n"
    "2. 每项必须给出 group:包含 strength(规格,如 20μg)、batch(批次,如有)、"
    "attribute(质量属性名,如 膜厚 / 总杂 / 含量)、table(该项所属表名,如 "
    "含量均匀度 / 有关物质检测结果,用于区分同名列并支撑专项判定)。\n"
    "3. 每项必须给出可在原文逐字定位的 source_ref(仅含该项自身的原文片段)。\n"
    "4. 规格限度单独成项(field 用 spec_limit),并在 group.attribute 标明其约束的属性。\n"
    "5. 严禁臆造原文中不存在的任何数值。\n\n"
    "【输出 JSON 结构】\n"
    "{{\n"
    '  "tables": [{{"table_type": "...", "title": "...", "source_document": "...", '
    '"is_stability_time_series": false, "columns": [], "sample_rows": []}}],\n'
    '  "intent": "descriptive_summary|shelf_life_extrapolation|compatibility|unknown",\n'
    '  "items": [\n'
    '    {{"field": "膜厚", "value": "0.05", "source_ref": "膜厚 0.05", '
    '"group": {{"strength": "20μg", "batch": "B1", "attribute": "膜厚", "table": "物理特性检测结果"}}}},\n'
    '    {{"field": "spec_limit", "value": "总杂≤2.0%", "source_ref": "限度 总杂≤2.0%", '
    '"group": {{"attribute": "总杂", "table": "有关物质检测结果"}}}}\n'
    "  ],\n"
    '  "spec_limits_present": true\n'
    "}}\n只输出 JSON,不要解释。"
)


class _LLMServiceAdapter:
    """把底座 ``svc.llm``(``LLMService.complete``)适配为 :class:`UnderstandingLLM`。"""

    def __init__(self, llm: Any) -> None:
        self._llm = llm

    def profile_and_extract(self, text: str, goal: str) -> Optional[dict]:
        complete = getattr(self._llm, "complete", None)
        if not callable(complete):
            return None
        # 截断超长文本,避免 token 超限(理解阶段无需全文)。
        max_chars = 8000
        clipped = text[:max_chars] + ("\n…[截断]" if len(text) > max_chars else "")
        user = _UNDERSTANDING_USER_TMPL.format(text=clipped, goal=goal or "(未提供)")
        try:
            result = complete(_UNDERSTANDING_SYSTEM, user, temperature=0.0, scope="understanding")
        except Exception as exc:  # noqa: BLE001 - LLM 调用异常视为不可用
            logger.info("理解层 LLM 调用失败:%s", exc)
            return None
        if not getattr(result, "ok", False):
            return None
        return _safe_parse_json(getattr(result, "content", "") or "")


def make_understanding_llm(svc: Any) -> Optional[UnderstandingLLM]:
    """构造理解层 LLM 适配器;无可用 LLM 时返回 ``None``(驱动澄清模式,需求 6.1)。

    判定「可用」的最小条件:``svc.llm`` 存在且具备可调用的 ``complete``。真实可用性
    (密钥 / 网络)在调用时体现——失败会被 :meth:`_LLMServiceAdapter.profile_and_extract`
    转为 ``None``,上层据此进入澄清。
    """
    llm = getattr(svc, "llm", None)
    if llm is None:
        return None
    if not callable(getattr(llm, "complete", None)):
        return None
    return _LLMServiceAdapter(llm)


def _safe_parse_json(content: str) -> Optional[dict]:
    """从 LLM 文本中安全解析 JSON 对象;失败返回 ``None``(不抛异常)。"""
    if not content:
        return None
    # 优先提取 ```json ... ``` 代码块。
    m = re.search(r"```(?:json)?\s*([\s\S]*?)\s*```", content)
    candidate = m.group(1) if m else content.strip()
    if not candidate.startswith("{"):
        # 退一步:截取首个 { 到末个 } 的范围。
        start = candidate.find("{")
        end = candidate.rfind("}")
        if start < 0 or end <= start:
            return None
        candidate = candidate[start : end + 1]
    try:
        parsed = json.loads(candidate)
    except (ValueError, TypeError):
        return None
    return parsed if isinstance(parsed, dict) else None


# ===========================================================================
# 任务 3:确定性防护(词典与归一助手)
# ===========================================================================

#: 保留时间 / RRT / 峰面积等**非稳定性时间**的列名词典(需求 2.6 / Property 3)。
_RETENTION_LEXICON = (
    "保留时间", "retention", "rrt", "峰面积", "peak area", "peakarea",
    "理论塔板", "拖尾因子", "分离度",
)

#: 表头 / 标签类单元格词典:这些是**列名 / 结构性文字**,不得当作测定值(需求 2.6)。
_HEADER_LABEL_LEXICON = (
    "样品名称", "名称", "限度", "结论", "项目", "批号", "编号",
    "接受标准", "标准", "规格", "时间", "time",
)


def is_retention_like_column(header: str) -> bool:
    """判断列名是否属于「保留时间 / RRT / 峰面积」等非稳定性时间语义(需求 2.6)。

    命中则其数值**永不**作为稳定性时间点,即使 LLM 误标也由本防护兜底。
    """
    h = (header or "").strip().lower()
    if not h:
        return False
    # 单位 "min" 仅在与保留时间语境共现时才算(避免误伤),但显式 min 列也排除作时间点。
    if h in ("min", "保留时间min", "rt", "rt(min)"):
        return True
    return any(kw in h for kw in _RETENTION_LEXICON)


def is_header_or_label_cell(cell: str, known_headers: Optional[list[str]] = None) -> bool:
    """判断单元格是否为表头 / 标签类文字(不得当作测定值,需求 2.6)。

    - 命中通用表头 / 标签词典;或
    - 等于本表已知列名之一(``known_headers``)。
    """
    c = (cell or "").strip()
    if not c:
        return False
    cl = c.lower()
    if known_headers:
        norm_headers = {str(h).strip().lower() for h in known_headers}
        if cl in norm_headers:
            return True
    return any(kw == cl or kw in c for kw in _HEADER_LABEL_LEXICON if kw)


def looks_like_stability_timepoints(values: list, headers: Optional[list[str]] = None) -> bool:
    """判断一组数值是否像合法的稳定性时间点序列(需求 2.6 防护)。

    合法时间点的确定性最小特征:
    - 至少 2 个数值;
    - 非负、且能解析为数值;
    - 单调非降(稳定性考察时间点递增);
    - 不全部相等(保留时间常彼此接近且非递增)。
    若列名命中保留时间词典,则**直接判否**(最强防护)。
    """
    if headers and any(is_retention_like_column(h) for h in headers):
        return False
    nums: list[float] = []
    for v in values:
        try:
            nums.append(float(str(v).strip()))
        except (TypeError, ValueError):
            return False
    if len(nums) < 2:
        return False
    if any(n < 0 for n in nums):
        return False
    if len(set(nums)) == 1:
        return False
    # 单调非降。
    if any(b < a for a, b in zip(nums, nums[1:])):
        return False
    return True


def no_silent_defaults(
    extracted: dict,
    missing_items: list[str],
) -> dict:
    """移除静默默认值:把缺失的规格 / 主 CQA / 目标时间点标记为缺失,绝不填默认。

    (需求 4.3 / Property 2)入参 ``extracted`` 为初步提取的字段映射;本函数原地补充
    ``missing_items`` 并返回**清洗后的** dict(不含 0.5 / 总杂质 / [24,36] 之类默认)。
    """
    cleaned = dict(extracted or {})

    # 规格限度:仅当确有来源时保留;否则标记缺失(不填 0.5)。
    spec = cleaned.get("spec_limit")
    if spec in (None, "", []) or not cleaned.get("spec_limit_source"):
        cleaned.pop("spec_limit", None)
        if "规格限度" not in missing_items:
            missing_items.append("规格限度")

    # 主 CQA:缺失即标记(不填 "总杂质")。
    if not cleaned.get("primary_cqa"):
        cleaned.pop("primary_cqa", None)
        if "主要质量指标" not in missing_items:
            missing_items.append("主要质量指标")

    # 目标时间点:缺失即标记(不填 [24, 36])。
    tps = cleaned.get("target_timepoints")
    if not tps:
        cleaned.pop("target_timepoints", None)
        if "目标时间点" not in missing_items:
            missing_items.append("目标时间点")

    return cleaned


__all__ = [
    "UnderstandingLLM",
    "make_understanding_llm",
    "is_retention_like_column",
    "is_header_or_label_cell",
    "looks_like_stability_timepoints",
    "no_silent_defaults",
    "DocumentProfiler",
    "IntentClassifier",
    "InformationExtractor",
    "BackReferenceValidator",
    "UnderstandingLayer",
    "harvest_observations",
    "build_intent_slots",
    "SLOT_PRIMARY_CQA",
    "SLOT_SPEC_LIMIT",
    "SLOT_TARGET_TIMEPOINTS",
    "SLOT_TARGET_ATTRIBUTES",
]


# ===========================================================================
# 任务 4:DocumentProfiler(多文档聚合)
# ===========================================================================

from kernel.task_sheet import (  # noqa: E402 - 置于助手之后,避免顶部循环
    CLARIFY_DATA_INTENT_MISMATCH,
    CLARIFY_INTENT_UNKNOWN,
    CLARIFY_NO_LLM,
    AnalysisTaskSheet,
    ClarificationDecision,
    DetectedTable,
    DocumentProfile,
    ExtractedItem,
    Intent,
    IntentSlot,
    SlotState,
    TableType,
)

#: 表类型关键词 → TableType(用于无 LLM / 校正 LLM 输出的确定性兜底)。
_TABLE_TYPE_LEXICON: tuple[tuple[TableType, tuple[str, ...]], ...] = (
    (TableType.RELATED_SUBSTANCES, ("有关物质", "杂质", "related substance", "impurit")),
    (TableType.CONTENT_UNIFORMITY, ("含量均匀度", "content uniformity", "uniformity")),
    (TableType.DISSOLUTION, ("溶出", "dissolution")),
    (TableType.ASSAY, ("含量", "assay", "potency", "效价")),
    (TableType.PHYSICAL_PROPERTIES, ("物理特性", "性状", "physical", "外观", "硬度", "脆碎")),
)


def _canon_table_type(title: str, columns: Optional[list[str]] = None) -> TableType:
    """按标题 / 列名关键词推断表类型(确定性,不依赖 LLM)。"""
    hay = (title or "") + " " + " ".join(columns or [])
    hay = hay.lower()
    for ttype, kws in _TABLE_TYPE_LEXICON:
        if any(kw in hay for kw in kws):
            return ttype
    return TableType.UNKNOWN


class DocumentProfiler:
    """文档画像器:识别表类型、是否稳定性时序,并跨文档聚合(需求 2)。"""

    def __init__(self, llm: Optional[UnderstandingLLM] = None) -> None:
        self._llm = llm

    def profile(
        self,
        text: str,
        file_names: Optional[list[str]] = None,
        *,
        goal: str = "",
    ) -> DocumentProfile:
        """产出 :class:`DocumentProfile`。

        优先使用 LLM 的表清单(若注入且可用),再用确定性防护**强制**校正:保留时间 /
        峰面积列绝不标为稳定性时序(需求 2.6 / Property 3)。无 LLM 时退回标题/列名
        关键词的确定性画像(仍受同一防护约束)。
        """
        file_names = file_names or []
        tables: list[DetectedTable] = []

        llm_out = self._llm.profile_and_extract(text, goal) if self._llm else None
        if llm_out and isinstance(llm_out.get("tables"), list):
            for raw in llm_out["tables"]:
                tables.append(self._build_table_from_llm(raw, file_names))
        else:
            tables.extend(self._heuristic_tables(text, file_names))

        # 确定性防护:保留时间 / 峰面积主导的表,强制 is_stability_time_series=False。
        for t in tables:
            if any(is_retention_like_column(c) for c in t.columns):
                t.is_stability_time_series = False

        has_stability = any(t.is_stability_time_series for t in tables)
        notes: list[str] = []
        if not has_stability:
            notes.append("未检测到稳定性时间序列数据。")
        return DocumentProfile(tables=tables, has_stability_time_series=has_stability, notes=notes)

    def _build_table_from_llm(self, raw: dict, file_names: list[str]) -> DetectedTable:
        raw = raw or {}
        title = str(raw.get("title", ""))
        columns = [str(c) for c in (raw.get("columns") or [])]
        # 表类型:信任 LLM,但用关键词兜底校正未知。
        ttype = TableType.UNKNOWN
        try:
            ttype = TableType(str(raw.get("table_type")))
        except (ValueError, TypeError):
            ttype = _canon_table_type(title, columns)
        if ttype is TableType.UNKNOWN:
            ttype = _canon_table_type(title, columns)
        src = str(raw.get("source_document") or (file_names[0] if file_names else ""))
        is_ts = bool(raw.get("is_stability_time_series", False))
        if ttype is TableType.STABILITY_TIME_SERIES:
            is_ts = True
        return DetectedTable(
            table_type=ttype,
            title=title,
            source_document=src,
            is_stability_time_series=is_ts,
            columns=columns,
            sample_rows=[[str(c) for c in row] for row in (raw.get("sample_rows") or [])],
        )

    def _heuristic_tables(self, text: str, file_names: list[str]) -> list[DetectedTable]:
        """无 LLM 时的确定性画像:按「=== File: name ===」分段 + 标题关键词识别表类型。

        这是一个**保守**的画像——它不试图解析数值(那是 compute 的事),只识别文本中
        出现了哪些类型的表,以支撑意图/澄清判定。
        """
        tables: list[DetectedTable] = []
        segments = _split_by_file(text, file_names)
        for src, body in segments:
            for title in _candidate_table_titles(body):
                ttype = _canon_table_type(title)
                if ttype is TableType.UNKNOWN:
                    continue
                tables.append(DetectedTable(
                    table_type=ttype, title=title.strip(), source_document=src,
                    is_stability_time_series=False,
                ))
        return tables


def _split_by_file(text: str, file_names: list[str]) -> list[tuple[str, str]]:
    """按 ``=== File: name ===`` 分隔符切分文本为 ``[(source, body)]``。"""
    if not text:
        return []
    parts = re.split(r"=== File:\s*(.*?)\s*===", text)
    if len(parts) <= 1:
        default_src = file_names[0] if file_names else ""
        return [(default_src, text)]
    out: list[tuple[str, str]] = []
    # re.split 结构:[前导, name1, body1, name2, body2, ...]
    i = 1
    while i < len(parts) - 1:
        out.append((parts[i].strip(), parts[i + 1]))
        i += 2
    return out


def _candidate_table_titles(body: str) -> list[str]:
    """启发式抽取可能的表标题行(短行、含已知表类型关键词)。"""
    titles: list[str] = []
    for line in (body or "").split("\n"):
        s = line.strip()
        if not s or len(s) > 40:
            continue
        if _canon_table_type(s) is not TableType.UNKNOWN:
            titles.append(s)
    return titles


# ===========================================================================
# 任务 5:IntentClassifier
# ===========================================================================

_DESCRIPTIVE_KW = ("梳理", "总结", "汇总", "规律", "概览", "概述", "整理", "describe", "summary", "summarize", "overview")
_SHELF_LIFE_KW = ("货架期", "有效期", "外推", "预测", "shelf life", "shelf-life", "extrapolat", "多少个月", "多少月")
_COMPATIBILITY_KW = (
    "相容性", "配伍", "相互作用", "相容", "compatibility", "compatible",
    "辅料筛选", "处方筛选", "excipient compatibility", "drug-excipient",
)


class IntentClassifier:
    """意图分类器:LLM 优先 + 确定性关键词兜底(需求 3)。"""

    def __init__(self, llm: Optional[UnderstandingLLM] = None) -> None:
        self._llm = llm

    def classify(
        self,
        goal: str,
        profile: DocumentProfile,
        *,
        llm_hint: Optional[str] = None,
        has_smiles: bool = False,
        has_excipient: bool = False,
    ) -> Intent:
        """返回 :class:`Intent`。

        - 若提供 ``llm_hint``(来自一次已发生的 LLM 调用),优先采用其可识别值。
        - **相容性强信号**:同时提供 SMILES 与辅料(相容性 Skill 的输入签名)→ 直接判定
          为 COMPATIBILITY(这是确定性输入信号,优先于易混的关键词)。
        - 否则用目标文本关键词判定;相容性 / 外推 / 描述关键词;都未命中则 UNKNOWN。
        """
        if llm_hint:
            try:
                hinted = Intent(str(llm_hint))
                # LLM 明确给出非 UNKNOWN 意图时采用;UNKNOWN 不应压制下方确定性强信号。
                if hinted is not Intent.UNKNOWN:
                    return hinted
            except (ValueError, TypeError):
                pass
        # 相容性输入签名(SMILES + 辅料):最强确定性信号。
        if has_smiles and has_excipient:
            return Intent.COMPATIBILITY
        g = (goal or "").lower()
        has_compat = any(kw.lower() in g for kw in _COMPATIBILITY_KW)
        has_desc = any(kw in g for kw in _DESCRIPTIVE_KW)
        has_shelf = any(kw in g for kw in _SHELF_LIFE_KW)
        # 相容性关键词优先于描述/外推(语义更专一)。
        if has_compat and not has_shelf:
            return Intent.COMPATIBILITY
        if has_shelf and not has_desc:
            return Intent.SHELF_LIFE
        if has_desc and not has_shelf:
            return Intent.DESCRIPTIVE_SUMMARY
        # 仅有 SMILES(无辅料)且无明确外推/描述意图 → 也倾向相容性。
        if has_smiles and not (has_shelf or has_desc):
            return Intent.COMPATIBILITY
        # 二者都命中或都未命中:交由澄清(需求 3.4)。
        return Intent.UNKNOWN


# ===========================================================================
# 任务 6:InformationExtractor + BackReferenceValidator
# ===========================================================================

# 确定性观测采集:列名词典(用于判定一行是否为含属性列名的表头)。
_NUM_TOKEN_RE = re.compile(r"-?\d+(?:\.\d+)?")
#: 删失/不等式型观测:前导比较符(半角/全角)+ 数值(+可选单位),如 >100、<0.05、≤0.1%。
#: 这类值是合法 QC 测定结果(耐折度>100、杂质<LOQ 等),不应被当作非数值而丢弃。
_CENSORED_OBS_RE = re.compile(
    r"^\s*(?:≤|≥|<=|>=|<=?|>=?|<|>)\s*-?\d+(?:\.\d+)?\s*[%a-zA-Zμ·/℃°]*$"
)
_STRENGTH_RE = re.compile(r"\d+\s*(?:μg|ug|mg|g)\b", re.IGNORECASE)
_BATCH_RE = re.compile(r"[A-Za-z]{2,}-?\d{4,}[-\w]*")


def _is_observation_value(s: str) -> bool:
    """单元格是否为可采集的观测值:纯数值,或删失/不等式型数值(如 >100、<0.05)。"""
    s = (s or "").strip()
    if not s:
        return False
    return bool(_NUM_TOKEN_RE.fullmatch(s) or _CENSORED_OBS_RE.match(s))


#: 统计/中间计算类列名(描述性梳理应排除,避免污染,需求 7.2)。
#: 注:「百分含量」是 assay 相对标示量结果列(关键 CQA,90~110% 限度的判定对象),
#: 不属于中间统计列,故**不**列入此词典(修复 assay 结果列被误丢)。
_HELPER_COLUMN_LEXICON = (
    "a+2.2s", "a+", "样品量", "含量均值", "相对偏差", "稀释倍数",
    "峰面积", "rsd", "标准偏差", "sd", "理论塔板", "拖尾",
)


def _is_helper_column(col: str) -> bool:
    """是否为统计/中间计算列(非真实质量属性)。"""
    c = (col or "").strip().lower()
    if not c:
        return True
    if c in ("a", "s"):  # 含量均匀度表的中间量 A / S
        return True
    return any(k in c for k in _HELPER_COLUMN_LEXICON)


def _header_attr_count(row: list) -> int:
    """行中"像属性名"的非数值单元格数量(用于识别表头行)。"""
    cnt = 0
    for c in row:
        if c is None:
            continue
        s = str(c).strip()
        if not s:
            continue
        if _NUM_TOKEN_RE.fullmatch(s):
            continue
        cnt += 1
    return cnt


def _row_strength(row: list) -> str:
    """从一行中提取规格 token(任一单元格命中 数字+μg/mg)。"""
    for c in row:
        if c is None:
            continue
        m = _STRENGTH_RE.search(str(c))
        if m:
            return m.group(0).replace(" ", "")
    return ""


def _row_batch(row: list) -> str:
    for c in row:
        if c is None:
            continue
        m = _BATCH_RE.search(str(c))
        if m:
            return m.group(0)
    return ""


def harvest_observations(profile: "DocumentProfile", tables: Optional[list] = None) -> list[ExtractedItem]:
    """从**结构化表格网格**确定性采集逐属性观测(修复列错位,需求 4 / 7.2)。

    ``tables`` 为 :class:`services.file_service.TableGrid` 列表(保留空单元格=None)。
    对每张表:定位属性表头行 → 其后数据行按**真实列索引**与表头属性配对 → 跳过保留时间 /
    统计辅助列。空单元格被保留为 None,列位置不漂移(这是相对"拍平文本"的关键修复)。

    数值直接来自文件,天然满足防捏造。无 ``tables`` 时返回空(上层据此走澄清 / 兜底)。
    """
    grids = tables or []
    items: list[ExtractedItem] = []
    observed_attrs: set[str] = set()
    attrs_with_spec: set[str] = set()
    for grid in grids:
        rows = getattr(grid, "rows", None) or (grid.get("rows") if isinstance(grid, dict) else None) or []
        title = getattr(grid, "title", "") or (grid.get("title", "") if isinstance(grid, dict) else "")
        # 定位表头行:第一行"属性名"非数值单元格 >= 2 且本身几乎不含纯数值。
        header_idx = -1
        for i, row in enumerate(rows):
            if _header_attr_count(row) >= 2:
                numeric = sum(1 for c in row if c is not None and _NUM_TOKEN_RE.fullmatch(str(c).strip()))
                if numeric == 0:
                    header_idx = i
                    break
        if header_idx < 0:
            continue
        header = [(str(c).strip() if c is not None else "") for c in rows[header_idx]]

        for row in rows[header_idx + 1:]:
            if not any(c is not None for c in row):
                continue
            label_cell = next((str(c) for c in row if c is not None), "")

            # —— 限度行:首列含"限度/接受标准/标准/规格",按列对齐抓规格限度 ——
            if _is_limit_row(label_cell):
                for idx, cell in enumerate(row):
                    if cell is None or idx >= len(header):
                        continue
                    s = str(cell).strip()
                    if not s or s in ("/", "—", "-", "/"):
                        continue
                    col = header[idx]
                    if not col or is_retention_like_column(col) or _is_helper_column(col):
                        continue
                    if is_header_or_label_cell(col) or not _looks_like_limit_value(s):
                        continue
                    attribute = _clean_header_attribute(col)
                    if not attribute:
                        continue
                    items.append(ExtractedItem(
                        field="spec_limit",
                        value=s,
                        source_ref=f"{label_cell} {col} {s}".strip(),
                        located=True,
                        group={"attribute": attribute, "table": title},
                        source="llm",
                    ))
                    attrs_with_spec.add(attribute)
                continue

            strength = _row_strength(row)
            batch = _row_batch(row)
            # 数据行需有 规格/批号 标识,否则可能是另一段表头或说明行。
            if not (strength or batch):
                continue
            for idx, cell in enumerate(row):
                if cell is None or idx >= len(header):
                    continue
                s = str(cell).strip()
                if not _is_observation_value(s):
                    continue
                col = header[idx]
                if not col or is_retention_like_column(col) or _is_helper_column(col):
                    continue
                if is_header_or_label_cell(col):
                    continue
                attribute = _clean_header_attribute(col)
                if not attribute:
                    continue
                observed_attrs.add(attribute)
                items.append(ExtractedItem(
                    field=attribute,
                    value=s,
                    source_ref=f"{label_cell} {col} {s}".strip(),
                    located=True,
                    group={"strength": strength, "batch": batch,
                           "attribute": attribute, "table": title},
                    source="llm",  # 来自文件确定性采集,等同可信来源
                ))

    # —— 保守的散文式限度解析(仅在属性名与操作符**字面相邻**时关联,否则放弃)——
    items.extend(_harvest_prose_limits(grids, observed_attrs, attrs_with_spec))
    return items


def _attr_core(attribute: str) -> str:
    """属性核心名:去掉尾部 % / 单位 / 空白,用于在散文中做字面匹配。"""
    s = re.sub(r"[%%\s]+$", "", str(attribute or "")).strip()
    s = re.sub(r"(mg/g|mg|g|s|mpa·s|mpa|cm|mm)$", "", s, flags=re.IGNORECASE).strip()
    return s


def _harvest_prose_limits(grids: list, observed_attrs: set, attrs_with_spec: set) -> list[ExtractedItem]:
    """从散文式限度文字中**保守**提取规格限度(需求 7.4,零误判取向)。

    规则(严格、宁可放弃也不误关联):
    - 只在某**已观测到的属性名**于一个子句中**紧邻**比较符 / 区间出现时才关联;
    - 已有列对齐限度的属性不再覆盖;
    - 形如"含右美托咪啶应为90~110%"(用 API 名而非属性名"含量")→ 字面不含"含量" → **放弃**;
    - "L=15.0""忽略限"等无明确属性归属 → **放弃**。
    """
    out: list[ExtractedItem] = []
    if not observed_attrs:
        return out
    # 属性核心名 → 原属性名(取最短核心,避免误配)。
    cores = {a: _attr_core(a) for a in observed_attrs}
    seen: set[str] = set(attrs_with_spec)

    for grid in grids:
        rows = getattr(grid, "rows", None) or (grid.get("rows") if isinstance(grid, dict) else None) or []
        title = getattr(grid, "title", "") or (grid.get("title", "") if isinstance(grid, dict) else "")
        for row in rows:
            for cell in row:
                if cell is None:
                    continue
                s = str(cell).strip()
                if not _has_limit_operator(s):
                    continue
                for clause in re.split(r"[;;。\n]", s):
                    clause = clause.strip()
                    if not clause or not _has_limit_operator(clause):
                        continue
                    for attr, core in cores.items():
                        if attr in seen or not core:
                            continue
                        if _attr_adjacent_to_limit(core, clause):
                            out.append(ExtractedItem(
                                field="spec_limit",
                                value=clause,
                                source_ref=clause,
                                located=True,
                                group={"attribute": attr, "table": title},
                                source="llm",
                            ))
                            seen.add(attr)
                    # —— assay 含量范围措辞:「含<活性成分>应为 90.0~110.0%」——
                    # 这是中文药典 assay 验收的通用表述(限度以活性成分名而非属性名
                    # 表述)。仅当文档确有 assay 百分含量列、且范围落在标示量量级
                    # (上界≥80)时关联,避免误绑水分/杂质等小范围限度。
                    assay_attr = _assay_percent_attr(observed_attrs)
                    if assay_attr and assay_attr not in seen:
                        am = _ASSAY_RANGE_RE.search(clause)
                        if am and float(am.group(2)) >= 80.0:
                            out.append(ExtractedItem(
                                field="spec_limit",
                                value=f"{am.group(1)}~{am.group(2)}%",
                                source_ref=clause,
                                located=True,
                                group={"attribute": assay_attr, "table": title},
                                source="llm",
                            ))
                            seen.add(assay_attr)
    return out


#: assay 含量范围措辞:「含<活性成分>(应)为 <lo>~<hi>%」(限度以 API 名表述)。
_ASSAY_RANGE_RE = re.compile(
    r"含[^,。;,;\s]{0,20}?(?:应为|应是|为)\s*(\d+(?:\.\d+)?)\s*[~~\--]\s*(\d+(?:\.\d+)?)\s*%?"
)


def _assay_percent_attr(observed_attrs) -> str:
    """在已观测属性中找 assay 相对标示量百分比列(优先「百分含量」,其次含%的「含量」)。"""
    for a in observed_attrs:
        if "百分含量" in str(a):
            return a
    for a in observed_attrs:
        if "含量" in str(a) and "%" in str(a):
            return a
    return ""


def _has_limit_operator(s: str) -> bool:
    return bool(re.search(r"[≤≥<><>]=?|~|~|不大于|不小于|nmt|nlt", str(s or ""), re.IGNORECASE))


def _attr_adjacent_to_limit(core: str, clause: str) -> bool:
    """属性核心名是否在子句中**紧邻**(其后 0–6 字符内)出现比较符/区间+数值。"""
    if core not in clause:
        return False
    pat = re.escape(core) + r".{0,6}?(?:[≤≥<><>]=?|~|~|不大于|不小于|nmt|nlt)\s*\d"
    return bool(re.search(pat, clause, re.IGNORECASE))


def _is_limit_row(label_cell: str) -> bool:
    """首列是否表明这是一条"限度 / 接受标准"行。"""
    c = (label_cell or "").strip().lower()
    return any(k in c for k in ("限度", "接受标准", "标准", "规格", "spec", "acceptance", "limit"))


def _looks_like_limit_value(s: str) -> bool:
    """单元格是否是一个**简洁的**限度表述(用于列对齐限度,排除长句散文)。

    仅接受:起始即比较符+数字(≤12.0 / <120s)、区间(2000~7000)、或纯数值。
    长句 / 散文(如"含右美托咪啶应为90~110%""应符合…规定(L=15.0)")一律拒绝,
    交由保守的散文匹配器按字面相邻关联(避免列对齐误绑)。
    """
    s = (s or "").strip()
    if not s or len(s) > 12:
        return False
    if re.match(r"^[≤≥<><>]=?\s*\d", s):
        return True
    if re.match(r"^\d+(?:\.\d+)?\s*[~~]\s*\d", s):
        return True
    if _NUM_TOKEN_RE.fullmatch(s):
        return True
    return False


def _clean_header_attribute(col: str) -> str:
    """清洗列名为属性名:去单位括号 / 斜杠单位后缀。"""
    s = re.sub(r"[//].*$", "", str(col or "")).strip()
    s = re.sub(r"[\((].*?[\))]", "", s).strip()
    return s


class InformationExtractor:
    """信息提取器:把 LLM 的 items 规整为 :class:`ExtractedItem` 列表(需求 4)。

    本层**不**判定是否可定位(那是 :class:`BackReferenceValidator` 的职责);它只负责
    结构化整形,并保留每项的 ``source_ref`` 原文片段。无 LLM 时返回空列表(上层据此
    判定缺失 / 进入澄清,绝不臆造)。
    """

    def __init__(self, llm: Optional[UnderstandingLLM] = None) -> None:
        self._llm = llm

    def extract(
        self,
        text: str,
        profile: DocumentProfile,
        intent: Intent,
        *,
        llm_out: Optional[dict] = None,
        goal: str = "",
        tables: Optional[list] = None,
    ) -> list[ExtractedItem]:
        out = llm_out
        if out is None and self._llm is not None:
            out = self._llm.profile_and_extract(text, goal)

        items: list[ExtractedItem] = []
        if out and isinstance(out.get("items"), list):
            for raw in out["items"]:
                if not isinstance(raw, dict):
                    continue
                field_name = str(raw.get("field", "")).strip()
                if not field_name:
                    continue
                value = str(raw.get("value", "")).strip()
                # 防护:表头 / 标签类文字不得作为测定值(需求 2.6)。
                if field_name == "value" and is_header_or_label_cell(value):
                    continue
                items.append(ExtractedItem(
                    field=field_name,
                    value=value,
                    source_ref=str(raw.get("source_ref", "")).strip(),
                    located=True,
                    group=dict(raw.get("group") or {}),
                ))

        # 结构化网格优先(关键架构修正):当文件解析出**结构化表格网格**时,以确定性
        # 网格采集为权威来源——它按真实列索引对齐、数值直取自单元格(最强防捏造),
        # 且统一覆盖删失值 / 限度行 / assay 措辞等。LLM 擅长「理解 / 意图 / 表分型」,
        # 不擅长「穷举逐格提取」,故有网格时不依赖 LLM 逐项枚举(修复 LLM 路径绕过
        # 确定性修复、且 LLM 常漏列 / 漏单位的问题)。
        if tables:
            harvested = harvest_observations(profile, tables)
            if harvested:
                return harvested
        # 无结构化网格(粘贴文本 / 不可解析文件)→ 回退 LLM 逐项提取(经回填校验防捏造)。
        return items


class BackReferenceValidator:
    """回填校验器:确认提取项的**数值确实存在于原文**,丢弃疑似捏造项(需求 4.1 / 4.2 / Property 1)。

    设计要点(修复「引文过严误杀」):防捏造的本质是「**数值必须来自原文**」,而非「引文
    必须逐字连续」。因此校验**以数值为中心**:
    1. 用户手动项(``source=="user_edited"``)豁免(责任在用户)。
    2. 提取项的数值 token 若出现在原文(归一空白后)→ 判定可定位、保留。
    3. 非数值项(如文字型规格 / 结论)退回到「``source_ref`` 子串」判定,保持原有严格度。
    4. 均不满足 → 视为疑似捏造,丢弃并记入 missing。
    """

    _NUM_RE = re.compile(r"-?\d+(?:\.\d+)?")

    @staticmethod
    def validate(items: list[ExtractedItem], text: str) -> tuple[list[ExtractedItem], list[str]]:
        """返回 ``(kept, missing_fields)``。"""
        haystack = text or ""
        norm_hay = re.sub(r"\s+", "", haystack)
        kept: list[ExtractedItem] = []
        missing: list[str] = []
        for it in items:
            # 1) 用户手动修改 / 新增项豁免(责任在用户),保留标记供审计区分。
            if getattr(it, "source", "llm") == "user_edited":
                it.located = True
                kept.append(it)
                continue

            if BackReferenceValidator._is_grounded(it, haystack, norm_hay):
                it.located = True
                kept.append(it)
            else:
                if it.field not in missing:
                    missing.append(it.field)
        return kept, missing

    @staticmethod
    def _is_grounded(it: ExtractedItem, haystack: str, norm_hay: str) -> bool:
        """判定单项是否「来自原文」:以数值为中心,文字项回退引文子串。"""
        value = (it.value or "").strip()
        nums = BackReferenceValidator._NUM_RE.findall(value)
        if nums:
            # 数值项:要求其**全部**数值 token 都能在原文找到(防止部分编造)。
            norm_nums_ok = all(
                BackReferenceValidator._number_in_text(n, haystack) for n in nums
            )
            if norm_nums_ok:
                return True
            # 数值未全部命中时,再给引文子串一次机会(容忍单位/格式差异)。
        # 文字型项(无数值,如文字规格 / 结论)或数值项兜底:引文子串判定。
        ref = re.sub(r"\s+", "", (it.source_ref or "").strip())
        if ref and ref in norm_hay:
            return True
        # 文字型且无引文:用 value 自身做子串判定(避免漏掉合法文字项)。
        if not nums:
            nv = re.sub(r"\s+", "", value)
            return bool(nv and nv in norm_hay)
        return False

    @staticmethod
    def _number_in_text(num: str, haystack: str) -> bool:
        """数值 token 是否出现在原文(允许尾随零差异,如 0.5 ↔ 0.50)。"""
        if not num:
            return False
        if num in haystack:
            return True
        # 容忍浮点格式差异:把原文所有数值规整后比对。
        try:
            target = float(num)
        except ValueError:
            return False
        for m in BackReferenceValidator._NUM_RE.findall(haystack):
            try:
                if float(m) == target:
                    return True
            except ValueError:
                continue
        return False


# ===========================================================================
# 意图保真:意图槽位派生(带来源态 / 置信度,禁止静默默认)
# ===========================================================================

#: 槽位名常量(对外契约,供澄清 / 路由 / 计算消费)。
SLOT_PRIMARY_CQA = "primary_cqa"
SLOT_SPEC_LIMIT = "spec_limit"
SLOT_TARGET_TIMEPOINTS = "target_timepoints"
#: 描述性梳理意图维度:用户希望梳理的目标质量属性集合(需求 15)。
SLOT_TARGET_ATTRIBUTES = "target_attributes"

#: 常见主 CQA 关键词(用户在指令中明示主指标时识别)。
_CQA_KEYWORDS = (
    "总杂", "单杂", "有关物质", "杂质", "含量", "溶出", "水分", "效价",
    "assay", "impurit", "dissolution", "potency", "moisture",
)

#: 目标时间点:匹配「24个月 / 36 月 / 24月」等中文月份表述。
_MONTHS_RE = re.compile(r"(\d+)\s*个?\s*月")


def _confidence_for(state: "SlotState") -> float:
    """来源态 → 默认置信度(STATED 最高、INFERRED 居中、MISSING 为 0)。"""
    return {SlotState.STATED: 1.0, SlotState.INFERRED: 0.6, SlotState.MISSING: 0.0}[state]


def build_intent_slots(
    goal: str,
    intent: Intent,
    extracted_items: Optional[list[ExtractedItem]] = None,
) -> list[IntentSlot]:
    """派生带保真元信息的意图槽位(需求 3 / 5;意图保真核心)。

    原则(确保用户真实意图被采纳):
    - 用户在 ``goal`` 中**明示** → ``STATED``(置信 1.0,最高优先)。
    - 仅能由提取项 / 文档**推断** → ``INFERRED``(置信居中,必须回显来源供改正)。
    - 既无明示也无可靠推断(或多候选歧义)→ ``MISSING``(触发选择性澄清,**绝不填默认**)。

    刻意**不产生任何默认值**(不填 0.5 / 总杂质 / [24,36])——缺失即 ``MISSING``,
    交由澄清回到用户。当前为货架期外推(``SHELF_LIFE``)与描述性梳理
    (``DESCRIPTIVE_SUMMARY``)派生关键槽位;其它意图返回空列表(不强加槽位,
    保持向后兼容)。
    """
    if intent is Intent.DESCRIPTIVE_SUMMARY:
        return _build_descriptive_slots(goal, extracted_items or [])
    if intent is not Intent.SHELF_LIFE:
        return []

    g = goal or ""
    items = extracted_items or []
    slots: list[IntentSlot] = []

    # —— 槽位 1:目标时间点(影响外推结论 → 缺失必问)——
    months = _MONTHS_RE.findall(g)
    if months:
        values = sorted({int(m) for m in months})
        slots.append(IntentSlot(
            name=SLOT_TARGET_TIMEPOINTS, value=values, state=SlotState.STATED,
            confidence=_confidence_for(SlotState.STATED),
            evidence=_first_match_span(_MONTHS_RE, g), affects_result=True,
        ))
    else:
        slots.append(IntentSlot(
            name=SLOT_TARGET_TIMEPOINTS, value=None, state=SlotState.MISSING,
            confidence=0.0, affects_result=True,
        ))

    # —— 槽位 2:主 CQA(影响外推对象 → 缺失 / 歧义必问)——
    cqa_in_goal = next((k for k in _CQA_KEYWORDS if k.lower() in g.lower()), "")
    if cqa_in_goal:
        slots.append(IntentSlot(
            name=SLOT_PRIMARY_CQA, value=cqa_in_goal, state=SlotState.STATED,
            confidence=_confidence_for(SlotState.STATED),
            evidence=cqa_in_goal, affects_result=True,
        ))
    else:
        candidates = _distinct_attributes(items)
        if len(candidates) == 1:
            slots.append(IntentSlot(
                name=SLOT_PRIMARY_CQA, value=candidates[0], state=SlotState.INFERRED,
                confidence=_confidence_for(SlotState.INFERRED),
                evidence=f"提取自文档属性:{candidates[0]}", affects_result=True,
            ))
        else:
            # 0 个 → 无依据;≥2 个 → 歧义。两者都不臆断,标记缺失待澄清。
            slots.append(IntentSlot(
                name=SLOT_PRIMARY_CQA, value=None, state=SlotState.MISSING,
                confidence=0.0, affects_result=True,
            ))

    # —— 槽位 3:规格限度(影响「合规」判定,但缺失可优雅降级为「不做合规判定」)——
    # affects_result=False:依信息价值原则,规格缺失不强制追问(避免过度澄清),
    # 仅以来源态透明呈现;下游据 spec_provided 决定是否做合规结论。
    spec_item = next((it for it in items if it.field == "spec_limit" and it.value), None)
    goal_spec = bool(re.search(r"[≤≥<><>]=?\s*\d|不大于|不小于|nmt|nlt", g, re.IGNORECASE))
    if goal_spec:
        slots.append(IntentSlot(
            name=SLOT_SPEC_LIMIT, value=None, state=SlotState.STATED,
            confidence=_confidence_for(SlotState.STATED),
            evidence="指令中含限度表述", affects_result=False,
        ))
    elif spec_item is not None:
        slots.append(IntentSlot(
            name=SLOT_SPEC_LIMIT, value=spec_item.value, state=SlotState.INFERRED,
            confidence=_confidence_for(SlotState.INFERRED),
            evidence=spec_item.source_ref or spec_item.value, affects_result=False,
        ))
    else:
        slots.append(IntentSlot(
            name=SLOT_SPEC_LIMIT, value=None, state=SlotState.MISSING,
            confidence=0.0, affects_result=False,
        ))

    return slots


def _build_descriptive_slots(goal: str, items: list[ExtractedItem]) -> list[IntentSlot]:
    """为描述性梳理(``DESCRIPTIVE_SUMMARY``)派生意图槽位(需求 15.1 / 15.3)。

    唯一关键意图维度:``target_attributes``——用户希望梳理哪些质量属性。
    - 指令**点名**质量属性(命中 ``_CQA_KEYWORDS``)→ ``STATED``,``value`` 为去重属性名列表。
    - 未点名 → ``MISSING``(``value=None``),但 ``affects_result=False``:描述性梳理未指定时
      默认梳理**全部**属性是良性、透明的默认(非篡改意图的静默默认),故不强制澄清,仅在确认
      面板透明呈现供用户可选收窄。**绝不臆造具体属性默认值**。
    """
    g = goal or ""
    gl = g.lower()
    mentioned: list[str] = []
    for kw in _CQA_KEYWORDS:
        if kw.lower() in gl and kw not in mentioned:
            mentioned.append(kw)
    if mentioned:
        return [IntentSlot(
            name=SLOT_TARGET_ATTRIBUTES, value=mentioned, state=SlotState.STATED,
            confidence=_confidence_for(SlotState.STATED),
            evidence="、".join(mentioned), affects_result=False,
        )]
    return [IntentSlot(
        name=SLOT_TARGET_ATTRIBUTES, value=None, state=SlotState.MISSING,
        confidence=0.0, affects_result=False,
    )]


def _distinct_attributes(items: list["ExtractedItem"]) -> list[str]:
    """从提取项收集去重的质量属性名(排除规格限度项与空值)。"""
    seen: list[str] = []
    for it in items:
        if it.field == "spec_limit":
            continue
        attr = str((it.group or {}).get("attribute") or "").strip()
        if attr and attr not in seen:
            seen.append(attr)
    return seen


def _first_match_span(pattern: re.Pattern, text: str) -> str:
    """返回首个匹配的原文片段(供槽位 evidence 溯源),无匹配返回空串。"""
    m = pattern.search(text or "")
    return m.group(0) if m else ""


# ===========================================================================
# 任务 7:UnderstandingLayer.understand 编排 + 澄清门禁
# ===========================================================================

#: 意图 → 建议 Skill id(与 Router.INTENT_TO_SKILL 保持一致)。
_INTENT_TO_SKILL = {
    Intent.DESCRIPTIVE_SUMMARY: "descriptive_summary",
    Intent.SHELF_LIFE: "stability",
    Intent.COMPATIBILITY: "compatibility",
}


class UnderstandingLayer:
    """理解层编排器:产出《分析任务单》或澄清任务单(需求 1 / 6)。"""

    def __init__(self, svc: Any) -> None:
        self.svc = svc

    def understand(self, raw: Any) -> AnalysisTaskSheet:
        """Phase-A 入口:纯编排、无分析数值计算(需求 1.4 / Property 4)。

        流程:无 LLM → 澄清;汇集文本;文档画像 → 意图分类 → 提取 → 回填校验;
        数据/意图不匹配 → 澄清;意图未知 → 澄清;任何异常 → 澄清(宁可多问)。
        """
        try:
            llm = make_understanding_llm(self.svc)
            if llm is None:
                return self._clarify(CLARIFY_NO_LLM, "understanding.clarify.no_llm")

            goal = str(getattr(raw, "goal", "") or "")
            file_names = [getattr(f, "name", "") or str(f) for f in (getattr(raw, "files", None) or [])]
            # 结构化摄入:保留表格网格(含空单元格),并拿到拍平文本供 LLM。
            text, grids = self._gather_structured(raw)

            # 一次 LLM 调用,结果在画像 / 意图 / 提取间复用(确定性、省 token)。
            llm_out = llm.profile_and_extract(text, goal)

            profiler = DocumentProfiler(llm)
            profile = self._profile_with_cached(profiler, text, file_names, goal, llm_out)

            # 确定性增强:转置 / 宽表稳定性数据(批次为列、时间点为行)LLM 常漏判。
            # 用结构化网格做确定性识别,命中则强制标记为稳定性时序,避免「未检测到稳定性
            # 时间序列数据」导致的货架期外推误澄清(修复转置宽表场景)。
            self._augment_stability_from_grids(profile, grids, file_names)

            classifier = IntentClassifier(llm)
            intent = classifier.classify(
                goal, profile,
                llm_hint=(llm_out or {}).get("intent") if llm_out else None,
                has_smiles=bool(str(getattr(raw, "smiles", "") or "").strip()),
                has_excipient=bool(str(getattr(raw, "excipient", "") or "").strip()),
            )

            extractor = InformationExtractor(llm)
            raw_items = extractor.extract(
                text, profile, intent, llm_out=llm_out, goal=goal, tables=grids
            )
            kept, missing = BackReferenceValidator.validate(raw_items, text)

            # 意图未知 → 澄清(需求 3.4 / 6.4)。
            if intent is Intent.UNKNOWN:
                sheet = self._clarify(CLARIFY_INTENT_UNKNOWN, "understanding.clarify.intent_unknown")
                sheet.document_profile = profile
                sheet.extracted_items = kept
                sheet.missing_items = missing
                self._audit("understanding_clarify", CLARIFY_INTENT_UNKNOWN)
                return sheet

            # 数据 / 意图不匹配:货架期外推但无稳定性时序 → 澄清(需求 6.3)。
            if intent is Intent.SHELF_LIFE and not profile.has_stability_time_series:
                sheet = self._clarify(
                    CLARIFY_DATA_INTENT_MISMATCH, "understanding.clarify.data_intent_mismatch"
                )
                sheet.intent = intent
                sheet.document_profile = profile
                sheet.extracted_items = kept
                sheet.missing_items = missing
                self._audit("understanding_clarify", CLARIFY_DATA_INTENT_MISMATCH)
                return sheet

            sheet = AnalysisTaskSheet(
                intent=intent,
                document_profile=profile,
                extracted_items=kept,
                proposed_skill_id=_INTENT_TO_SKILL.get(intent, ""),
                missing_items=missing,
                source_method="llm",
                intent_slots=build_intent_slots(goal, intent, kept),
            )
            self._audit("understanding_built", intent.value)
            return sheet
        except Exception as exc:  # noqa: BLE001 - 任何异常转澄清,绝不硬算(需求 6)
            logger.warning("理解层处理异常,转入澄清:%s", exc, exc_info=True)
            return self._clarify(CLARIFY_INTENT_UNKNOWN, "understanding.clarify.error")

    # ------------------------------------------------------------------
    @staticmethod
    def _augment_stability_from_grids(
        profile: DocumentProfile, grids: list, file_names: list[str]
    ) -> None:
        """用结构化网格确定性识别转置/宽表稳定性时序,命中则增强 profile(原地)。

        转置宽表(批次为列、时间点为行,如「0天/加速1月/加速6月」)LLM 与启发式都常漏判。
        本方法用 :func:`skills.stability.grid_parser.parse_stability_grids` 做确定性识别;
        命中则设 ``has_stability_time_series=True``、补一张稳定性表、清除「未检测到」提示。
        任何异常安全忽略(不改变原 profile)。
        """
        if profile.has_stability_time_series or not grids:
            return
        try:
            from skills.stability.grid_parser import parse_stability_grids

            data = parse_stability_grids(grids)
        except Exception:  # noqa: BLE001 - 解析不可用时不增强
            return
        if not data or not data.get("batches"):
            return
        src = file_names[0] if file_names else ""
        cqa = str(data.get("primary_cqa", "") or "稳定性")
        profile.tables.append(DetectedTable(
            table_type=TableType.STABILITY_TIME_SERIES,
            title=cqa,
            source_document=src,
            is_stability_time_series=True,
            columns=[cqa],
        ))
        profile.has_stability_time_series = True
        profile.notes = [n for n in (profile.notes or []) if "未检测到稳定性" not in n]

    # ------------------------------------------------------------------
    @staticmethod
    def _profile_with_cached(
        profiler: "DocumentProfiler",
        text: str,
        file_names: list[str],
        goal: str,
        llm_out: Optional[dict],
    ) -> DocumentProfile:
        """复用已有的 llm_out 构建画像,避免二次 LLM 调用。"""
        tables: list[DetectedTable] = []
        if llm_out and isinstance(llm_out.get("tables"), list):
            for raw_t in llm_out["tables"]:
                tables.append(profiler._build_table_from_llm(raw_t, file_names))
        else:
            tables.extend(profiler._heuristic_tables(text, file_names))
        for t in tables:
            if any(is_retention_like_column(c) for c in t.columns):
                t.is_stability_time_series = False
        has_stability = any(t.is_stability_time_series for t in tables)
        notes: list[str] = [] if has_stability else ["未检测到稳定性时间序列数据。"]
        return DocumentProfile(tables=tables, has_stability_time_series=has_stability, notes=notes)

    def _gather_text(self, raw: Any) -> str:
        """汇集上传文件文本(经 ``svc.file``)与直接粘贴文本。失败安全降级。"""
        text, _ = self._gather_structured(raw)
        return text

    def _gather_structured(self, raw: Any):
        """汇集为 ``(flattened_text, table_grids)``。

        优先用 ``svc.file.parse_structured`` 拿到保留空单元格的表格网格(供确定性采集
        按真实列索引配对);同时拼接拍平文本供 LLM。``parse_structured`` 不可用时退回
        ``parse_text``(仅文本、无网格),保证向后兼容与稳健。
        """
        file_svc = getattr(self.svc, "file", None)
        parse_structured = getattr(file_svc, "parse_structured", None)
        parse_text = getattr(file_svc, "parse_text", None)

        parts: list[str] = []
        grids: list = []
        for f in (getattr(raw, "files", None) or []):
            name = getattr(f, "name", None) or str(f)
            try:
                path, cleanup = _materialize_file(f)
                try:
                    if callable(parse_structured):
                        doc = parse_structured(path)
                        for g in getattr(doc, "tables", None) or []:
                            grids.append(g)
                        content = getattr(doc, "text", "") or ""
                    elif callable(parse_text):
                        content = parse_text(path) or ""
                    else:
                        content = ""
                finally:
                    cleanup()
            except Exception as exc:  # noqa: BLE001
                logger.info("理解层结构化解析文件失败(%s):%s", name, exc)
                content = ""
            if content:
                parts.append(f"\n=== File: {name} ===\n{content}\n")

        pasted = (getattr(raw, "extra", None) or {}).get("text_content")
        if pasted:
            parts.append(str(pasted))
        return "".join(parts), grids

    def _clarify(self, reason_code: str, message_key: str) -> AnalysisTaskSheet:
        return AnalysisTaskSheet(
            intent=Intent.UNKNOWN,
            clarification=ClarificationDecision(
                needed=True,
                reason_code=reason_code,
                message_key=message_key,
                options=["descriptive_summary", "shelf_life_extrapolation"],
            ),
            source_method="clarification",
        )

    def _audit(self, event: str, detail: str) -> None:
        audit = getattr(self.svc, "audit", None)
        recorder = getattr(audit, "record_event", None) or getattr(audit, "record_analysis", None)
        if callable(recorder):
            try:
                recorder("understanding", event, detail)
            except Exception:  # noqa: BLE001 - 审计失败不影响主流程
                logger.debug("理解层审计记录失败(已忽略)。", exc_info=True)


def _materialize_file(f):
    """把上传对象 / 路径解析为可被 FileService 读取的磁盘路径,返回 ``(path, cleanup)``。"""
    import os
    import tempfile

    def _noop() -> None:
        return None

    if isinstance(f, (str, os.PathLike)):
        p = str(f)
        if os.path.exists(p):
            return p, _noop

    name = getattr(f, "name", "") or "upload"
    suffix = os.path.splitext(name)[1] or ""
    data = None
    getvalue = getattr(f, "getvalue", None)
    if callable(getvalue):
        data = getvalue()
    else:
        read = getattr(f, "read", None)
        if callable(read):
            try:
                f.seek(0)
            except Exception:  # noqa: BLE001
                pass
            data = read()
    if data is None:
        return str(f), _noop
    if isinstance(data, str):
        data = data.encode("utf-8", errors="ignore")
    fd, tmp_path = tempfile.mkstemp(suffix=suffix)
    try:
        with os.fdopen(fd, "wb") as fh:
            fh.write(data)
    except Exception:  # noqa: BLE001
        return str(f), _noop

    def _cleanup() -> None:
        try:
            os.remove(tmp_path)
        except OSError:
            pass

    return tmp_path, _cleanup