File size: 31,004 Bytes
300df0f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
"""
Core cross-reference extractor.
"""
from __future__ import annotations

import re
import json
import logging
from pathlib import Path
from typing import Optional

from .models import (
    InternalRef, ExternalRef, ModificationRef,
    ExtractionResult, DocType, ModAction,
)

logger = logging.getLogger(__name__)


# ===========================================================================
# Regex catalogue
# ===========================================================================

# ── Internal references ─────────────────────────────────────────────────────
_RE_DIEU = r"[ĐĐð][iíì]ều\s+(\d+[a-zđ]?)(?!\w)"

_INTERNAL_PATTERNS: list[tuple[str, re.Pattern]] = [
    ("diem_khoan_dieu", re.compile(r"điểm\s+([a-zđ])\s+khoản\s+(\d+)\s+" + _RE_DIEU, re.IGNORECASE | re.UNICODE)),
    ("khoan_dieu", re.compile(r"khoản\s+(\d+)\s+" + _RE_DIEU, re.IGNORECASE | re.UNICODE)),
    ("dieu", re.compile(_RE_DIEU, re.IGNORECASE | re.UNICODE)),
]

# ── External references ─────────────────────────────────────────────────────
_EXTERNAL_PATTERNS: list[tuple[DocType, re.Pattern]] = [
    (DocType.LUAT, re.compile(r"(?:Bộ\s+)?[Ll]uật\s+[\w\s]+?số\s+(\d{1,3}/\d{4}/QH\d{1,2})", re.UNICODE)),
    (DocType.NGHI_DINH, re.compile(r"[Nn]ghị\s+đ[iị]nh\s+(?:số\s+)?(\d{1,3}/\d{4}/NĐ-CP)", re.UNICODE)),
    (DocType.TTLT, re.compile(r"[Tt]hông\s+tư\s+li[eê]n\s+t[ịi]ch\s+(?:số\s+)?(\d{1,3}/\d{4}/TTLT-[\w-]+)", re.UNICODE)),
    (DocType.THONG_TU, re.compile(r"[Tt]hông\s+tư\s+(?:số\s+)?(\d{1,3}/\d{4}/TT-[\w]+)", re.UNICODE)),
]

# ── Modification patterns ───────────────────────────────────────────────────
_MOD_ACTION_MAP: list[tuple[ModAction, re.Pattern]] = [
    (ModAction.THAY_THE, re.compile(r"[Tt]hay\s+thế", re.UNICODE)),
    (ModAction.BAI_BO, re.compile(r"[Bb]ãi\s+bỏ", re.UNICODE)),
    (ModAction.BO_SUNG, re.compile(r"[Bb]ổ\s+sung", re.UNICODE)),
    (ModAction.HET_HIEU_LUC, re.compile(r"hết\s+hiệu\s+lực", re.UNICODE | re.IGNORECASE)),
    (ModAction.SUA_DOI, re.compile(r"[Ss]ửa\s+đổi", re.UNICODE)),
]

_MOD_TARGET_PATTERN = re.compile(
    r"(?:(?:điểm|đpcm)\s+(?P<point>[a-zđ])\s+(?:vào\s+)?)??"
    r"(?:khoản\s+(?P<khoan>\d+[a-z]*)\s+)??"
    r"[Đđ][iíì]ều\s+(?P<dieu>\d+[a-zđ]?)(?!\w)"
    r"(?:\s+[\w\s]+?(?:số\s+(?P<skh>\S+)))?",
    re.UNICODE | re.IGNORECASE,
)

# Matches "vào sau Điều X" — the anchor article for insertion (bo_sung)
_RE_VAO_SAU = re.compile(
    r"vào\s+sau\s+[Đđ][iíì]ều\s+(?P<dieu>\d+[a-zđ]?)(?!\w)",
    re.UNICODE | re.IGNORECASE,
)

# Quoted content — should NOT be scanned for relationships.
# Covers: "straight ASCII", \u201c curved \u201d, and mixed open/close variants.
# Also handles the common Vietnamese legal pattern: ": " ... "" (opened with straight, closed with curved)
_OPEN_QUOTES  = '"\u201c\u2018\u2019'   # ", ", ', '
_CLOSE_QUOTES = '"\u201d\u2018\u2019'   # ", ", ', '
_RE_QUOTED = re.compile(
    r'[' + _OPEN_QUOTES + r'][^' + _CLOSE_QUOTES + r']{0,3000}?[' + _CLOSE_QUOTES + r']',
    re.DOTALL | re.UNICODE,
)


_RE_PREAMBLE_ANCHOR = re.compile(
    r"sửa\s+đổi,\s+bổ\s+sung\s+một\s+số\s+điều\s+của\s+([^,;]+?)\s+số\s+(\d+/\d+/[A-ZĐ-]+\d*)",
    re.IGNORECASE | re.UNICODE
)

_NEW_TEXT_PATTERN = re.compile(r"như\s+sau\s*:\s*['\"]?(.*?)['\"]?$", re.DOTALL | re.UNICODE)


class CrossReferenceExtractor:
    def __init__(
        self,
        lookup_table: dict[str, str],
        *,
        fuzzy_enabled: bool = True,
        short_title_map_path: Optional[str | Path] = "data/short_title_mapping.json",
    ) -> None:
        self._lookup: dict[str, str] = lookup_table
        self._fuzzy_enabled = fuzzy_enabled
        self._short_title_map: dict[str, str] = {}
        
        if short_title_map_path and Path(short_title_map_path).exists():
            try:
                with open(short_title_map_path, encoding="utf-8") as f:
                    self._short_title_map = json.load(f)
            except Exception as e:
                logger.warning("Failed to load short title map: %s", e)

    def _resolve_self_references(self, text: str, article_uid: str, clause_uid: Optional[str] = None, point_uid: Optional[str] = None) -> str:
        if not text:
            return text
            
        # Parse current indices from UIDs (format: doc_123_dieu_5_khoan_2_diem_a)
        def get_idx(uid, marker):
            if not uid: return None
            parts = uid.split('_')
            try:
                idx = parts.index(marker)
                return parts[idx+1]
            except ValueError:
                return None
                
        curr_art = get_idx(article_uid, 'dieu')
        curr_clause = get_idx(clause_uid, 'khoan') if clause_uid else None
        curr_point = get_idx(point_uid, 'diem') if point_uid else None
        
        # 1. Resolve "Điều này" -> "Điều X"
        if curr_art:
            text = re.sub(r'(?i)\bĐiều\s+này\b', f'Điều {curr_art}', text)
            
        # 2. Resolve "Khoản này" -> "Khoản Y Điều X"
        if curr_clause and curr_art:
            text = re.sub(r'(?i)\bkhoản\s+này\b', f'khoản {curr_clause} Điều {curr_art}', text)
            
        # 3. Resolve "Điểm này" -> "Điểm Z Khoản Y Điều X"
        if curr_point and curr_clause and curr_art:
            text = re.sub(r'(?i)\bđiểm\s+này\b', f'điểm {curr_point} khoản {curr_clause} Điều {curr_art}', text)
            
        return text

    def _expand_coordinate_chains(self, text: str) -> str:
        """
        Phase 1 Expansion:
        Handles cases like: "khoản 1, khoản 2 Điều 3" -> "khoản 1 Điều 3, khoản 2 Điều 3"
        This is a heuristic regex expansion to help the main extractor catch all items in a list.
        """
        if not text:
            return text
            
        # Pattern: (khoản X) (, hoặc "và") (khoản Y Điều Z)
        # Matches: khoản 1, khoản 2 Điều 3
        # Group 1: khoản 1
        # Group 2: , 
        # Group 3: khoản 2
        # Group 4: Điều 3
        pattern = r'(?i)(khoản\s+\d+[a-z]*)\s*(,|và)\s*(khoản\s+\d+[a-z]*)\s+(Điều\s+\d+[a-zđ]?)(?!\w)'
        
        # We run it a few times in case of "khoản 1, khoản 2, khoản 3 Điều 4"
        for _ in range(3):
            new_text = re.sub(pattern, r'\1 \4 \2 \3 \4', text)
            if new_text == text:
                break
            text = new_text
            
        # Same for points: "điểm a, điểm b khoản 1" -> "điểm a khoản 1, điểm b khoản 1"
        pt_pattern = r'(?i)(điểm\s+[a-zđ])\s*(,|và)\s*(điểm\s+[a-zđ])\s+(khoản\s+\d+[a-z]*)'
        for _ in range(3):
            new_text = re.sub(pt_pattern, r'\1 \4 \2 \3 \4', text)
            if new_text == text:
                break
            text = new_text
            
        return text

    def extract_from_article(
        self,
        doc_id: str,
        article_uid: str,
        article_text: str,
        *,
        clause_uid: Optional[str] = None,
        point_uid: Optional[str] = None,
        is_modifying_doc: bool = False,
    ) -> ExtractionResult:
        result = ExtractionResult(doc_id=doc_id)
        
        # --- PHASE 1: Entity Recognition & Resolution ---
        # 1. Resolve self pronouns (Điều này, khoản này)
        resolved_text = self._resolve_self_references(article_text, article_uid, clause_uid, point_uid)
        
        # 2. Expand coordinate chains (khoản 1, khoản 2 Điều 3)
        resolved_text = self._expand_coordinate_chains(resolved_text)
        
        fragments = self._preprocess_text(resolved_text, is_modifying_doc)

        for fragment in fragments:
            occupied_spans: list[tuple[int, int]] = []

            if is_modifying_doc:
                try:
                    mods = self._extract_modifications(doc_id, article_uid, fragment)
                    result.modification_refs.extend(mods)
                    for m in mods:
                        occupied_spans.append((m.start_char, m.end_char))
                except Exception as exc:
                    result.parse_errors.append(f"modification [{article_uid}]: {exc}")

            try:
                internals, granular_externals, unified_mods = self._extract_unified_references(doc_id, article_uid, fragment, clause_uid, point_uid)
                
                for mod in unified_mods:
                    if not any(mod.start_char >= s and mod.end_char <= e for s, e in occupied_spans):
                        result.modification_refs.append(mod)
                        occupied_spans.append((mod.start_char, mod.end_char))

                for internal in internals:
                    if not any(internal.start_char >= s and internal.end_char <= e for s, e in occupied_spans):
                        result.internal_refs.append(internal)
                        occupied_spans.append((internal.start_char, internal.end_char))
                        
                for ext in granular_externals:
                    if not any(ext.start_char >= s and ext.end_char <= e for s, e in occupied_spans):
                        result.external_refs.append(ext)
                        occupied_spans.append((ext.start_char, ext.end_char))
            except Exception as exc:
                result.parse_errors.append(f"unified_refs [{article_uid}]: {exc}")
                
            # try:
            #     externals = self._extract_external(doc_id, article_uid, fragment, clause_uid, point_uid)
            #     for ext in externals:
            #         if not any(ext.start_char >= s and ext.end_char <= e for s, e in occupied_spans):
            #             result.external_refs.append(ext)
            #             occupied_spans.append((ext.start_char, ext.end_char))
            # except Exception as exc:
            #     result.parse_errors.append(f"external_standalone [{article_uid}]: {exc}")

        return result

    def _preprocess_text(self, text: str, is_modifying_doc: bool) -> list[str]:
        if not text: return []
        text = re.sub(r"\s*/\s*", "/", text)
        text = " ".join(text.split())
        if is_modifying_doc:
            # RULE 1: Strip everything after "như sau: <open-quote>" 
            # — the text after that is the NEW inserted content, NOT a reference.
            # This handles the case where the closing quote is missing (truncated segment).
            _RE_NHU_SAU_OPEN = re.compile(
                r'(như\s+sau\s*:?\s*)["' + '\u201c\u2018' + r'].*$',
                re.DOTALL | re.UNICODE | re.IGNORECASE,
            )
            text_for_scan = _RE_NHU_SAU_OPEN.sub(r'\1"…"', text)

            # RULE 2: Also strip fully-closed quoted blocks (e.g. "..." or "...")
            text_for_scan = _RE_QUOTED.sub('"…"', text_for_scan)

            raw_fragments = [f.strip() for f in text_for_scan.split(";") if f.strip()]
            final_fragments = []
            temp = ""
            for i, frag in enumerate(raw_fragments):
                if "điều" in frag.lower() or i == len(raw_fragments) - 1:
                    final_fragments.append((temp + " " + frag).strip())
                    temp = ""
                else:
                    temp += " " + frag
            return final_fragments
        return [text]

    def resolve_external(self, ref: ExternalRef) -> ExternalRef:
        # Danh sách các ứng viên để thử tra cứu (Ưu tiên Short Title trước)
        candidates = []
        if ref.raw_so_ky_hieu in self._short_title_map:
            candidates.append((self._short_title_map[ref.raw_so_ky_hieu], "short_title_map"))
        candidates.append((ref.raw_so_ky_hieu, "exact"))

        last_normalized = None
        for raw_val, method in candidates:
            logger.info("Resolving external: %s", raw_val)
            normalized = _normalize_so_ky_hieu(raw_val, ref.target_doc_type)
            if not last_normalized:
                last_normalized = normalized # Giữ lại bản chuẩn hóa của chuỗi gốc
            
            if normalized in self._lookup:
                ref.normalized_so_ky_hieu = normalized
                ref.target_doc_id = self._lookup[normalized]
                ref.match_method = method
                ref.confidence = 1.0
                return ref

        # Nếu không tìm thấy chính xác, lưu lại bản chuẩn hóa cuối cùng
        ref.normalized_so_ky_hieu = last_normalized

        # --- FUZZY MATCHING (Tạm thời tắt để tăng tốc độ) ---
        # if self._fuzzy_enabled:
        #     best, dist = _fuzzy_levenshtein(last_normalized, self._lookup)
        #     if dist <= 2:
        #         ref.target_doc_id = self._lookup[best]
        #         ref.match_method = "fuzzy_levenshtein"
        #         ref.confidence = max(0.0, 1.0 - dist * 0.15)
        
        return ref


    def _compile_unified_regex(self):
        if hasattr(self, '_unified_re'): return self._unified_re
        titles = [re.escape(k) for k in self._short_title_map.keys() if len(k) > 5]
        titles.sort(key=len, reverse=True)
        titles_pattern = "|".join(titles) if titles else "NOT_A_MATCH"
        
        pattern = (
            r"(?:[Đđ]iểm\s+(?P<point>[a-zđ])\s+)?"
            r"(?:[Kk]hoản\s+(?P<clause>\d+[a-z]*)\s+)?"
            r"[Đđ]iều\s+(?P<article>\d+[a-zđ]?)(?!\w)"
            r"(?:\s+(?:của\s+)?(?P<doc_ref>này|" + titles_pattern + r"|(?:Luật|Bộ luật|Nghị định|Thông tư liên tịch|Thông tư)\s+(?:số\s+)?\d{1,3}/\d{4}/\S+))?"
        )
        self._unified_re = re.compile(pattern, re.UNICODE)
        return self._unified_re

    def _extract_unified_references(self, doc_id, article_uid, text, clause_uid, point_uid):
        internals = []
        externals = []
        unified_mods = []
        seen = set()
        pattern = self._compile_unified_regex()
        
        for match in pattern.finditer(text):
            gd = match.groupdict()
            article = gd.get('article')
            clause = gd.get('clause')
            point = gd.get('point')
            doc_ref = gd.get('doc_ref')
            
            # Deduplicate exact same references in the same fragment
            key = (article, clause, point, doc_ref)
            if key in seen: continue
            seen.add(key)
            
            # --- PHASE 2: RELATION CLASSIFICATION ---
            # Quét ngược 60 ký tự (khoảng 10 từ) trước từ được tìm thấy
            lookback_text = text[max(0, match.start() - 60): match.start()].lower()
            
            # 1. Phát hiện quan hệ Ngoại trừ (Exception)
            is_exception = False
            if "trừ" in lookback_text or "ngoại trừ" in lookback_text or "không áp dụng" in lookback_text:
                is_exception = True
                
            # 2. Phát hiện hành vi Sửa đổi/Bổ sung
            is_mod = False
            action = ModAction.SUA_DOI
            
            # Danh sách từ khóa hành động và các từ chỉ định "bị động/tham chiếu"
            mod_keywords = ["sửa đổi", "bổ sung", "thay thế", "bãi bỏ", "hết hiệu lực"]
            passive_markers = ["được ", "đã ", "nêu tại", "theo ", "tại ", "quy định ", "thông tư ", "luật ", "nghị định "]
            negative_phrases = ["khai bổ sung", "tờ khai bổ sung", "mẫu biểu bổ sung"]
            
            # Kiểm tra xem có nằm trong cụm từ loại trừ không (ví dụ: "khai bổ sung")
            is_negative = any(np in lookback_text for np in negative_phrases)
            
            # Tìm từ khóa xuất hiện cuối cùng trong lookback (gần trích dẫn nhất)
            found_kw = None
            kw_pos = -1
            for kw in mod_keywords:
                pos = lookback_text.rfind(kw)
                if pos > kw_pos:
                    kw_pos = pos
                    found_kw = kw
            
            if found_kw and not is_negative:
                # Kiểm tra 20 ký tự ngay trước từ khóa đó để xem có phải bị động không
                context_before = lookback_text[max(0, kw_pos - 20): kw_pos]
                
                # Nếu không chứa các từ bị động, hoặc là bắt đầu một chỉ dẫn (đầu dòng/sau dấu chấm)
                is_passive = any(m in context_before for m in passive_markers)
                # Chú ý: "1. Sửa đổi" -> context_before là "1. " -> không passive
                is_start = context_before.strip() == "" or context_before.strip().endswith(".") or context_before.strip().endswith(":")
                
                if not is_passive or is_start:
                    is_mod = True
                    if "sửa đổi" == found_kw: action = ModAction.SUA_DOI
                    elif "bổ sung" == found_kw: action = ModAction.BO_SUNG
                    elif "thay thế" == found_kw: action = ModAction.THAY_THE
                    elif "bãi bỏ" == found_kw: action = ModAction.BAI_BO
                    elif "hết hiệu lực" == found_kw: action = ModAction.HET_HIEU_LUC
            
            if is_mod:
                # Trích xuất Clause hiện tại làm source
                source_cl = None
                if clause_uid:
                    parts = clause_uid.split('_')
                    if 'khoan' in parts:
                        source_cl = parts[parts.index('khoan') + 1]
                        
                target_skh = doc_ref if doc_ref and doc_ref.lower() != "này" else ""
                
                mod_ref = ModificationRef(
                    source_doc_id=doc_id, source_article_uid=article_uid,
                    source_clause_index=source_cl,
                    action=action,
                    raw_target_so_ky_hieu=target_skh,
                    target_article_index=article,
                    target_clause_index=clause,
                    target_point_label=point,
                    context_text=match.group(0),
                    start_char=match.start(),
                    end_char=match.end()
                )
                unified_mods.append(mod_ref)
                continue
            
            # 3. Mặc định là Internal/External Ref (Áp dụng, Căn cứ, Trích dẫn)
            if not doc_ref or doc_ref.lower() == "này":
                internals.append(InternalRef(
                    source_doc_id=doc_id, source_article_uid=article_uid,
                    source_clause_uid=clause_uid, source_point_uid=point_uid,
                    target_article_index=article, target_clause_index=clause, target_point_label=point,
                    context_text=match.group(0), start_char=match.start(), end_char=match.end(),
                    is_exception=is_exception
                ))
            else:
                doc_type = DocType.UNKNOWN
                if "Luật" in doc_ref or "Bộ luật" in doc_ref: doc_type = DocType.LUAT
                elif "Nghị định" in doc_ref: doc_type = DocType.NGHI_DINH
                elif "Thông tư liên tịch" in doc_ref: doc_type = DocType.TTLT
                elif "Thông tư" in doc_ref: doc_type = DocType.THONG_TU
                
                ref = ExternalRef(
                    source_doc_id=doc_id, source_article_uid=article_uid, source_clause_uid=clause_uid,
                    source_point_uid=point_uid, raw_so_ky_hieu=doc_ref, target_doc_type=doc_type,
                    target_article_index=article, target_clause_index=clause, target_point_label=point,
                    context_text=match.group(0), start_char=match.start(), end_char=match.end(),
                    is_exception=is_exception
                )
                self.resolve_external(ref)
                externals.append(ref)
                
        return internals, externals, unified_mods

    def _extract_internal(self, doc_id, article_uid, text, clause_uid, point_uid) -> list[InternalRef]:
        refs = []; seen = set()
        for p_name, pattern in _INTERNAL_PATTERNS:
            for match in pattern.finditer(text):
                groups = match.groups()
                tp = tc = ta = None
                if p_name == "diem_khoan_dieu": tp, tc, ta = groups
                elif p_name == "khoan_dieu": tc, ta = groups
                elif p_name == "dieu": ta = groups[0]
                key = (ta, tc, tp)
                if key not in seen:
                    refs.append(InternalRef(
                        source_doc_id=doc_id, source_article_uid=article_uid,
                        source_clause_uid=clause_uid, source_point_uid=point_uid,
                        target_article_index=ta, target_clause_index=tc, target_point_label=tp,
                        context_text=match.group(0), start_char=match.start(), end_char=match.end()
                    ))
                    seen.add(key)
        return refs

    def _extract_preamble_anchor(self, preamble_text: str) -> Optional[ExternalRef]:
        if not preamble_text: return None
        start_keywords = ["ban hành", "quy định chi tiết", "hướng dẫn"]
        search_area = preamble_text
        for kw in start_keywords:
            idx = preamble_text.lower().find(kw)
            if idx != -1:
                search_area = preamble_text[idx:]
                break
        boundary = search_area.lower().find("đã được")
        if boundary != -1:
            temp_area = search_area[:boundary]
            if _RE_PREAMBLE_ANCHOR.search(temp_area):
                search_area = temp_area
        match = _RE_PREAMBLE_ANCHOR.search(search_area)
        if not match: match = _RE_PREAMBLE_ANCHOR.search(preamble_text)
        if match:
            ref = ExternalRef(source_doc_id="", source_article_uid="", raw_so_ky_hieu=match.group(2).strip(),
                              target_doc_type=DocType.LUAT, context_text=match.group(0))
            self.resolve_external(ref)
            return ref
        return None

    def _extract_external(self, doc_id, article_uid, text, clause_uid, point_uid) -> list[ExternalRef]:
        refs = []
        for title in self._short_title_map:
            if title in text:
                start_idx = text.find(title)
                ref = ExternalRef(
                    source_doc_id=doc_id, source_article_uid=article_uid, source_clause_uid=clause_uid,
                    source_point_uid=point_uid, raw_so_ky_hieu=title,
                    target_doc_type=DocType.LUAT if "Luật" in title else DocType.NGHI_DINH,
                    context_text=text[max(0, start_idx-20):start_idx+len(title)+50],
                    start_char=start_idx, end_char=start_idx+len(title)
                )
                self.resolve_external(ref)
                refs.append(ref)
        for d_type, pattern in _EXTERNAL_PATTERNS:
            for match in pattern.finditer(text):
                ref = ExternalRef(
                    source_doc_id=doc_id, source_article_uid=article_uid, source_clause_uid=clause_uid,
                    source_point_uid=point_uid, raw_so_ky_hieu=match.group(1), target_doc_type=d_type,
                    context_text=match.group(0), start_char=match.start(), end_char=match.end()
                )
                self.resolve_external(ref)
                refs.append(ref)
        return refs

    def _extract_modifications(self, doc_id, article_uid, text) -> list[ModificationRef]:
        refs = []

        # RULE: Collect "vào sau Điều X" positions so we skip them as standalone refs.
        # E.g. "Bổ sung Điều 37a vào sau Điều 37" → only one ref pointing to Điều 37.
        vao_sau_positions: set[tuple[int,int]] = set()
        for vs in _RE_VAO_SAU.finditer(text):
            vao_sau_positions.add((vs.start(), vs.end()))

        full_matches = list(_MOD_TARGET_PATTERN.finditer(text))

        # Build a set of character-start positions for Điều that are NEWLY NAMED
        # (i.e., followed immediately by "vào sau Điều X").
        # E.g. "Bổ sung Điều 37a vào sau Điều 37": 37a is the new article name → skip.
        # The real target is the Điều inside the "vào sau" span.
        _RE_AFTER_DIEU = re.compile(
            r"[\s,;]+(?:[\w\s]+?\s+)?vào\s+sau\s+[Đđ][iíì]ều",
            re.UNICODE | re.IGNORECASE,
        )

        def _is_new_article_name(m: re.Match) -> bool:
            """Return True if this Điều match is the NEW article name in 'Bổ sung Điều X vào sau Điều Y'."""
            after = text[m.end(): m.end() + 80]
            return bool(_RE_AFTER_DIEU.match(after))

        bare_pattern = re.compile(
            r"(?:điểm\s+(?P<point>[a-zđ])\s+)??"
            r"khoản\s+(?P<khoan>\d+)",
            re.UNICODE | re.IGNORECASE,
        )
        bare_matches = [
            bm for bm in bare_pattern.finditer(text)
            if not any(bm.start() >= fm.start() and bm.end() <= fm.end() for fm in full_matches)
        ]
        all_matches = sorted(full_matches + bare_matches, key=lambda x: x.start())

        last_dieu = last_skh = None
        # Seed last_dieu from the first "real target" full match (not a new-article-name)
        for m in full_matches:
            if not _is_new_article_name(m):
                last_dieu, last_skh = m.group("dieu"), m.group("skh")
                break
        # Also check vao_sau targets as seed (they are the real destination)
        if not last_dieu and vao_sau_positions:
            first_vs = _RE_VAO_SAU.search(text)
            if first_vs:
                last_dieu = first_vs.group("dieu")

        for match in all_matches:
            gd = match.groupdict()

            # If this Điều match is the NEW article name (e.g. "Điều 37a" before "vào sau Điều 37")
            # → skip emitting a ref; the real ref will come from the "vào sau" target below.
            if gd.get("dieu") and _is_new_article_name(match):
                continue

            # For "vào sau Điều X": the Điều inside this span IS the real target
            if gd.get("dieu"):
                last_dieu = gd.get("dieu")
                if gd.get("skh"):
                    last_skh = gd.get("skh")


            # Determine action from text before this match
            local_action = ModAction.SUA_DOI
            pre_text = text[:match.start()]
            for act_type, pattern in reversed(_MOD_ACTION_MAP):
                if pattern.search(pre_text):
                    local_action = act_type
                    break

            # Source Clause detection: look back for numbered list items (e.g. "1.", "2.")
            # but exclude numbers that follow "Điều" (those are article numbers)
            source_clause = None
            clause_search = []
            for cm in re.finditer(r'(?:Khoản\s+)?(\d+)\.(?!\d)', pre_text):
                lookback = pre_text[max(0, cm.start() - 10): cm.start()].lower()
                if "điều" not in lookback:
                    clause_search.append(cm.group(1))
            if clause_search:
                source_clause = clause_search[-1]

            target_article = gd.get("dieu") or last_dieu
            is_partial = (target_article is None) and (gd.get("khoan") or gd.get("point"))

            ref = ModificationRef(
                source_doc_id=doc_id,
                source_article_uid=article_uid,
                source_clause_index=source_clause,
                action=local_action,
                raw_target_so_ky_hieu=gd.get("skh") or last_skh or "",
                target_article_index=target_article,
                target_clause_index=gd.get("khoan"),
                target_point_label=gd.get("point"),
                context_text=text,
                start_char=match.start(),
                end_char=match.end(),
                is_partial_ref=is_partial
            )
            if ref.raw_target_so_ky_hieu:
                temp = ExternalRef(
                    source_doc_id=doc_id, source_article_uid=article_uid,
                    raw_so_ky_hieu=ref.raw_target_so_ky_hieu,
                    target_doc_type=DocType.LUAT, context_text="",
                )
                self.resolve_external(temp)
                ref.target_doc_id = temp.target_doc_id
            refs.append(ref)

        # Deduplication logic: Remove general refs if a more specific one exists for the same target
        final_refs = []
        for i, r1 in enumerate(refs):
            is_redundant = False
            for j, r2 in enumerate(refs):
                if i == j: continue
                
                # Same target doc and article?
                same_doc = (r1.target_doc_id == r2.target_doc_id) if (r1.target_doc_id and r2.target_doc_id) else (r1.raw_target_so_ky_hieu == r2.raw_target_so_ky_hieu)
                if same_doc and r1.target_article_index == r2.target_article_index:
                    # R2 is strictly more specific?
                    if not r1.target_clause_index and r2.target_clause_index:
                        is_redundant = True; break
                    if (r1.target_clause_index and r1.target_clause_index == r2.target_clause_index and
                        not r1.target_point_label and r2.target_point_label):
                        is_redundant = True; break
            if not is_redundant:
                final_refs.append(r1)
        return final_refs


from src.data_pipeline.normalize import normalize as _normalize_a

def _normalize_so_ky_hieu(raw: str, doc_type: DocType) -> str:
    # Bản đồ chuyển đổi DocType (B) -> loai_van_ban (A)
    type_map = {
        DocType.LUAT: "Luật",
        DocType.BO_LUAT: "Bộ luật",
        DocType.NGHI_DINH: "Nghị định",
        DocType.THONG_TU: "Thông tư",
        DocType.TTLT: "Thông tư liên tịch"
    }
    loai_vb = type_map.get(doc_type, "")
    return _normalize_a(raw, loai_vb) or ""

def _fuzzy_levenshtein(query, lookup):
    if not lookup: return "", 999
    def _lev(a, b):
        if len(a) < len(b): return _lev(b, a)
        if not b: return len(a)
        prev = list(range(len(b) + 1))
        for i, ca in enumerate(a):
            curr = [i + 1]
            for j, cb in enumerate(b): curr.append(min(prev[j + 1] + 1, curr[j] + 1, prev[j] + (ca != cb)))
            prev = curr
        return prev[-1]
    best_key = min(lookup.keys(), key=lambda k: _lev(query, k))
    return best_key, _lev(query, best_key)