File size: 6,083 Bytes
17b6ec9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
# core_numeric.py
# 纯逻辑函数:判定“数字一致性”,辅助行筛选/归一化等;供 app.py 调用。

import re, unicodedata
from collections import Counter
import pandas as pd
import cn2an
from number_parser import parse as en_words_parse

# ===== 公共常量 =====
SUFFIX = "_numeric_tagged"
CATEGORY_TITLES = {
    "inconsistency in target", "inconsistency in source", "tag mismatch",
    "numeric mismatch", "target same as source",
}
PROCESS_ONLY_CATEGORIES = {"numeric mismatch"}

# ===== 正则与工具 =====
DIGIT_RE = re.compile(r'[-+]?\d+(?:[.,]\d+)?(?:[%%‰])?')
TAG_RE   = re.compile(r'<[^>]+>')
FILE_SEG_RE = re.compile(r'^(.*?)(?:\s*\((\d+)\))\s*$')
_WS = ("\u00A0", "\u2009", "\u2002", "\u2003", "\u2007", "\u202F")

MONTH_MAP = {"january":"1","jan":"1","february":"2","feb":"2","march":"3","mar":"3",
    "april":"4","apr":"4","may":"5","june":"6","jun":"6","july":"7","jul":"7",
    "august":"8","aug":"8","september":"9","sept":"9","sep":"9",
    "october":"10","oct":"10","november":"11","nov":"11","december":"12","dec":"12"}
MONTH_RE = re.compile(r'\b(' + '|'.join(sorted(MONTH_MAP.keys(), key=len, reverse=True)) + r')\b', re.I)
DAY_RE   = re.compile(r'\b(\d{1,2})(?:st|nd|rd|th)?\b', re.I)
YEAR_RE  = re.compile(r'\b(19|20)\d{2}\b')
EN_ORDINAL_MAP = {"first":"1","second":"2","third":"3","fourth":"4","fifth":"5",
    "sixth":"6","seventh":"7","eighth":"8","ninth":"9","tenth":"10",
    "eleventh":"11","twelfth":"12","thirteenth":"13","fourteenth":"14",
    "fifteenth":"15","sixteenth":"16","seventeenth":"17","eighteenth":"18",
    "nineteenth":"19","twentieth":"20"}
EN_ORDINAL_RE = re.compile(r'\b(' + '|'.join(EN_ORDINAL_MAP.keys()) + r')\b', re.I)
DOUBLE_SEMANTIC_PAT = re.compile(r'(二重|二層|双重|双层|ダブル|이중)')
VAR_TOKEN_PAT = re.compile(r'\b[A-Z]\d+\b')
FULLWIDTH_TO_HALF = str.maketrans("0123456789", "0123456789")

def clean_invis_spaces(s: str) -> str:
    if not isinstance(s, str): return s
    for ch in _WS: s = s.replace(ch, " ")
    return s

def strip_tags(s: str) -> str:
    return TAG_RE.sub("", s)

def normalize_text(s: str) -> str:
    if not isinstance(s, str): return ""
    s = unicodedata.normalize("NFKC", s).replace(",", ",").replace(".", ".").replace("%", "%")
    s = strip_tags(s)
    # 英文常见词数化
    try: s = en_words_parse(s)
    except Exception: pass
    # 中文数字转阿拉伯
    try: s = cn2an.transform(s, "cn2an")
    except Exception: pass
    return s

def add_month_tokens_if_date_context(text: str, counter: Counter):
    if not text: return
    for m in MONTH_RE.finditer(text):
        month_num = MONTH_MAP[m.group(1).lower()]
        pre  = text[max(0, m.start()-8): m.start()]
        post = text[m.end(): m.end()+8]
        if DAY_RE.search(pre) or DAY_RE.search(post) or YEAR_RE.search(post):
            counter.update([month_num])

def _to_ascii_num(s: str) -> str: return s.translate(FULLWIDTH_TO_HALF)

def normalize_cjk_dates_in_counter(text: str, counter: Counter):
    # 粗略把全角月/日换成半角,以便计数统一
    m = re.search(r'([0-9\d]{1,2})\s*月\s*([0-9\d]{1,2})\s*日', text)
    if m:
        mm = str(int(_to_ascii_num(m.group(1)))); dd = str(int(_to_ascii_num(m.group(2))))
        if counter.get(mm, 0) == 0: counter.update([mm])
        if counter.get(dd, 0) == 0: counter.update([dd])

def extract_numbers(s: str) -> Counter:
    s = re.sub(r'(\d)\s*[-–—~〜~]\s*(\d)', r'\1 \2', s)
    s = re.sub(r'(?<=\d)(?=[A-Za-z])', ' ', s)
    s = re.sub(r'(?<=[A-Za-z])(?=\d)', ' ', s)
    s = clean_invis_spaces(s)
    nums = [token.replace(",", "") for token in DIGIT_RE.findall(s) if token]
    counter = Counter(nums)
    add_month_tokens_if_date_context(s, counter)
    normalize_cjk_dates_in_counter(s, counter)
    return counter

def counter_to_str(c: Counter) -> str:
    if not c: return ""
    items = sorted(c.items(), key=lambda kv: kv[0])
    return ", ".join([f"{k}×{v}" if v>1 else k for k,v in items])

def balanced_mask_vars(src: str, tgt: str):
    src_set = set(VAR_TOKEN_PAT.findall(src))
    tgt_set = set(VAR_TOKEN_PAT.findall(tgt))
    if not src_set or src_set != tgt_set: return src, tgt, []
    def _repl(m: "re.Match") -> str: return m.group(0)[0] + "§VAR§"
    return VAR_TOKEN_PAT.sub(_repl, src), VAR_TOKEN_PAT.sub(_repl, tgt), sorted(src_set)

def classify_row(src: str, tgt: str):
    note = []
    s, t, vars_hits = balanced_mask_vars(src or "", tgt or "")
    if vars_hits: note.append("MaskedVars(" + ",".join(vars_hits) + ")")
    s_norm, t_norm = normalize_text(s), normalize_text(t)
    ns, nt = extract_numbers(s_norm), extract_numbers(t_norm)
    src_s, tgt_s = counter_to_str(ns), counter_to_str(nt)
    if not ns and not nt:
        base = "NoNumbersBothSides"
        return (False, src_s, tgt_s, base if not note else base+";"+";".join(note))
    if ns == nt:
        base = "NumbersEqualAfterNormalization"
        return (False, src_s, tgt_s, base if not note else base+";"+";".join(note))
    base = "NumbersDiffer"
    return (True, src_s, tgt_s, base if not note else base+";"+";".join(note))

def _is_blank(x) -> bool:
    if x is None or (isinstance(x, float) and pd.isna(x)): return True
    s = str(x).strip().lower()
    return s in {"", "nan", "none", "null"}

def is_category_header_row(a, c, d) -> bool:
    a_norm = clean_invis_spaces(str(a)).strip().lower()
    return (a_norm != "") and _is_blank(c) and _is_blank(d)

def is_allowed_category_name(a) -> bool:
    a_norm = clean_invis_spaces(str(a)).strip().lower()
    return any(a_norm.startswith(cat) for cat in PROCESS_ONLY_CATEGORIES)

def find_header_row_by_probe(get_cell, max_scan=60) -> int:
    """
    通过 get_cell(r, c) 探测第 C、D 列是否出现 'Source'/'Target'。
    适配 xlrd 行读取或 pandas.DataFrame.iat。
    """
    for r in range(max_scan):
        c = str(get_cell(r, 2) or "").strip().lower()
        d = str(get_cell(r, 3) or "").strip().lower()
        if c == "source" and d == "target":
            return r
    return -1