Spaces:
Runtime error
Runtime error
File size: 6,083 Bytes
17b6ec9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 | # core_numeric.py
# 纯逻辑函数:判定“数字一致性”,辅助行筛选/归一化等;供 app.py 调用。
import re, unicodedata
from collections import Counter
import pandas as pd
import cn2an
from number_parser import parse as en_words_parse
# ===== 公共常量 =====
SUFFIX = "_numeric_tagged"
CATEGORY_TITLES = {
"inconsistency in target", "inconsistency in source", "tag mismatch",
"numeric mismatch", "target same as source",
}
PROCESS_ONLY_CATEGORIES = {"numeric mismatch"}
# ===== 正则与工具 =====
DIGIT_RE = re.compile(r'[-+]?\d+(?:[.,]\d+)?(?:[%%‰])?')
TAG_RE = re.compile(r'<[^>]+>')
FILE_SEG_RE = re.compile(r'^(.*?)(?:\s*\((\d+)\))\s*$')
_WS = ("\u00A0", "\u2009", "\u2002", "\u2003", "\u2007", "\u202F")
MONTH_MAP = {"january":"1","jan":"1","february":"2","feb":"2","march":"3","mar":"3",
"april":"4","apr":"4","may":"5","june":"6","jun":"6","july":"7","jul":"7",
"august":"8","aug":"8","september":"9","sept":"9","sep":"9",
"october":"10","oct":"10","november":"11","nov":"11","december":"12","dec":"12"}
MONTH_RE = re.compile(r'\b(' + '|'.join(sorted(MONTH_MAP.keys(), key=len, reverse=True)) + r')\b', re.I)
DAY_RE = re.compile(r'\b(\d{1,2})(?:st|nd|rd|th)?\b', re.I)
YEAR_RE = re.compile(r'\b(19|20)\d{2}\b')
EN_ORDINAL_MAP = {"first":"1","second":"2","third":"3","fourth":"4","fifth":"5",
"sixth":"6","seventh":"7","eighth":"8","ninth":"9","tenth":"10",
"eleventh":"11","twelfth":"12","thirteenth":"13","fourteenth":"14",
"fifteenth":"15","sixteenth":"16","seventeenth":"17","eighteenth":"18",
"nineteenth":"19","twentieth":"20"}
EN_ORDINAL_RE = re.compile(r'\b(' + '|'.join(EN_ORDINAL_MAP.keys()) + r')\b', re.I)
DOUBLE_SEMANTIC_PAT = re.compile(r'(二重|二層|双重|双层|ダブル|이중)')
VAR_TOKEN_PAT = re.compile(r'\b[A-Z]\d+\b')
FULLWIDTH_TO_HALF = str.maketrans("0123456789", "0123456789")
def clean_invis_spaces(s: str) -> str:
if not isinstance(s, str): return s
for ch in _WS: s = s.replace(ch, " ")
return s
def strip_tags(s: str) -> str:
return TAG_RE.sub("", s)
def normalize_text(s: str) -> str:
if not isinstance(s, str): return ""
s = unicodedata.normalize("NFKC", s).replace(",", ",").replace(".", ".").replace("%", "%")
s = strip_tags(s)
# 英文常见词数化
try: s = en_words_parse(s)
except Exception: pass
# 中文数字转阿拉伯
try: s = cn2an.transform(s, "cn2an")
except Exception: pass
return s
def add_month_tokens_if_date_context(text: str, counter: Counter):
if not text: return
for m in MONTH_RE.finditer(text):
month_num = MONTH_MAP[m.group(1).lower()]
pre = text[max(0, m.start()-8): m.start()]
post = text[m.end(): m.end()+8]
if DAY_RE.search(pre) or DAY_RE.search(post) or YEAR_RE.search(post):
counter.update([month_num])
def _to_ascii_num(s: str) -> str: return s.translate(FULLWIDTH_TO_HALF)
def normalize_cjk_dates_in_counter(text: str, counter: Counter):
# 粗略把全角月/日换成半角,以便计数统一
m = re.search(r'([0-9\d]{1,2})\s*月\s*([0-9\d]{1,2})\s*日', text)
if m:
mm = str(int(_to_ascii_num(m.group(1)))); dd = str(int(_to_ascii_num(m.group(2))))
if counter.get(mm, 0) == 0: counter.update([mm])
if counter.get(dd, 0) == 0: counter.update([dd])
def extract_numbers(s: str) -> Counter:
s = re.sub(r'(\d)\s*[-–—~〜~]\s*(\d)', r'\1 \2', s)
s = re.sub(r'(?<=\d)(?=[A-Za-z])', ' ', s)
s = re.sub(r'(?<=[A-Za-z])(?=\d)', ' ', s)
s = clean_invis_spaces(s)
nums = [token.replace(",", "") for token in DIGIT_RE.findall(s) if token]
counter = Counter(nums)
add_month_tokens_if_date_context(s, counter)
normalize_cjk_dates_in_counter(s, counter)
return counter
def counter_to_str(c: Counter) -> str:
if not c: return ""
items = sorted(c.items(), key=lambda kv: kv[0])
return ", ".join([f"{k}×{v}" if v>1 else k for k,v in items])
def balanced_mask_vars(src: str, tgt: str):
src_set = set(VAR_TOKEN_PAT.findall(src))
tgt_set = set(VAR_TOKEN_PAT.findall(tgt))
if not src_set or src_set != tgt_set: return src, tgt, []
def _repl(m: "re.Match") -> str: return m.group(0)[0] + "§VAR§"
return VAR_TOKEN_PAT.sub(_repl, src), VAR_TOKEN_PAT.sub(_repl, tgt), sorted(src_set)
def classify_row(src: str, tgt: str):
note = []
s, t, vars_hits = balanced_mask_vars(src or "", tgt or "")
if vars_hits: note.append("MaskedVars(" + ",".join(vars_hits) + ")")
s_norm, t_norm = normalize_text(s), normalize_text(t)
ns, nt = extract_numbers(s_norm), extract_numbers(t_norm)
src_s, tgt_s = counter_to_str(ns), counter_to_str(nt)
if not ns and not nt:
base = "NoNumbersBothSides"
return (False, src_s, tgt_s, base if not note else base+";"+";".join(note))
if ns == nt:
base = "NumbersEqualAfterNormalization"
return (False, src_s, tgt_s, base if not note else base+";"+";".join(note))
base = "NumbersDiffer"
return (True, src_s, tgt_s, base if not note else base+";"+";".join(note))
def _is_blank(x) -> bool:
if x is None or (isinstance(x, float) and pd.isna(x)): return True
s = str(x).strip().lower()
return s in {"", "nan", "none", "null"}
def is_category_header_row(a, c, d) -> bool:
a_norm = clean_invis_spaces(str(a)).strip().lower()
return (a_norm != "") and _is_blank(c) and _is_blank(d)
def is_allowed_category_name(a) -> bool:
a_norm = clean_invis_spaces(str(a)).strip().lower()
return any(a_norm.startswith(cat) for cat in PROCESS_ONLY_CATEGORIES)
def find_header_row_by_probe(get_cell, max_scan=60) -> int:
"""
通过 get_cell(r, c) 探测第 C、D 列是否出现 'Source'/'Target'。
适配 xlrd 行读取或 pandas.DataFrame.iat。
"""
for r in range(max_scan):
c = str(get_cell(r, 2) or "").strip().lower()
d = str(get_cell(r, 3) or "").strip().lower()
if c == "source" and d == "target":
return r
return -1
|