File size: 8,650 Bytes
03b56f8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
"""Self-checking evaluation for the generic noisy-romaji rescue.

This is deliberately *generative*: recovery cases are produced by applying
corruption operators to clean compositions of known dictionary/colloquial
units, so the suite proves the mechanism generalizes rather than memorizing one
string. It also asserts the confidence gate *abstains* on heavy/ambiguous noise.

Corruptions are split by whether they preserve the intended *reading*:

* reading-preserving (style flip wapuro<->Hepburn, IME small-tsu spelling) are
  hard-asserted to recover exactly -- these are the realistic, high-value wins;
* reading-altering (triardupling, geminate drop, vowel inflation) are reported
  only, because a corrupted form may legitimately collide with another real word
  (e.g. ``itan`` -> 異端 vs ``ittan`` -> 一旦) and forcing one answer would be the
  very overfitting we are avoiding.

Run:

    python src/eval_general_phrase.py

Exits non-zero if any hard assertion fails.
"""

import sys
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parent))

from normalization import normalize_input
from romaji_kana import load_general_lexicon
from general_phrase import (
    build_general_phrase_index,
    canon_style,
    canonicalize_romaji_variants,
    general_phrase_rescue,
    _collapse_runs,
    _geminate_sokuon,
)

DEFAULT_GENERAL_LEXICON = "artifacts/lexicon/general_reading_lexicon.json"

PASS = []
FAIL = []


def check(name, cond, detail=""):
    (PASS if cond else FAIL).append(name)
    mark = "ok  " if cond else "FAIL"
    line = f"[{mark}] {name}"
    if detail:
        line += f"  ::  {detail}"
    print(line)


def report(name, value):
    print(f"[rpt ] {name}  ::  {value}")


# --------------------------------------------------------------------------- #
# Corruption operators (deterministic; no RNG)
# --------------------------------------------------------------------------- #

def op_style_flip(s):
    # wapuro -> Hepburn-ish flips that canon_style must absorb (reading-preserving)
    for a, b in (("sy", "sh"), ("ty", "ch"), ("tu", "tsu"), ("hu", "fu"), ("zy", "j")):
        if a in s:
            return s.replace(a, b, 1)
    return s


def op_sokuon_ime(s):
    # a geminate consonant -> IME small-tsu spelling (reading-preserving)
    for i in range(len(s) - 1):
        if s[i] == s[i + 1] and s[i] in "bcdfghjkmpqrstvwyz":
            return s[:i] + "xtu" + s[i + 1:]
    return s


def op_run(s):
    # triple the first consonant/glide (may alter reading near geminates)
    for i, ch in enumerate(s):
        if ch in "ybcdfghjkmpqrstvwz":
            return s[:i] + ch * 3 + s[i + 1:]
    return s


def op_drop_geminate(s):
    for i in range(len(s) - 1):
        if s[i] == s[i + 1] and s[i] in "bcdfghjkmpqrstvwyz":
            return s[:i] + s[i + 1:]
    return s


def op_long_vowel_inflate(s):
    for i, ch in enumerate(s):
        if ch in "aiueo":
            return s[:i] + ch + ch + s[i:]
    return s


# --------------------------------------------------------------------------- #

def main():
    g = load_general_lexicon(DEFAULT_GENERAL_LEXICON)
    index = build_general_phrase_index(g)

    print("=== canonicalizer unit tests ===")
    check("canon sh->sy", canon_style("shudan") == "syudan", canon_style("shudan"))
    check("canon sy stable", canon_style("syudan") == "syudan")
    check("canon shuuryou->syuuryou", canon_style("shuuryou") == "syuuryou", canon_style("shuuryou"))
    check("canon tsu->tu", canon_style("tsunami") == "tunami", canon_style("tsunami"))
    check("canon chi->ti", canon_style("chizu") == "tizu", canon_style("chizu"))
    check("canon fu->hu", canon_style("fujisan") == "huzisan", canon_style("fujisan"))
    check("canon ji->zi", canon_style("jisho") == "zisyo", canon_style("jisho"))
    check("sokuon moxtute->motte", _geminate_sokuon("moxtute") == "motte", _geminate_sokuon("moxtute"))
    check("sokuon ixtukai->ikkai", _geminate_sokuon("ixtukai") == "ikkai", _geminate_sokuon("ixtukai"))
    check("sokuon kixtute->kitte", _geminate_sokuon("kixtute") == "kitte", _geminate_sokuon("kixtute"))
    check("run syyyuu->syuu", _collapse_runs("syyyuu") == "syuu", _collapse_runs("syyyuu"))
    check("run aaaa->aa (vowel keeps 2)", _collapse_runs("aaaa") == "aa", _collapse_runs("aaaa"))
    check("run uu stable", _collapse_runs("uu") == "uu")
    variants = canonicalize_romaji_variants(
        "imamoxtutewruusyudannowataraitannsyyyuuryoudemoiiyo"
    )
    check("variants fold motte+syuu",
          any("motte" in v and "syuu" in v for v in variants),
          " | ".join(variants))
    check("variants repair tewruu only as additive candidate",
          any("imamotteru" in v for v in variants),
          " | ".join(variants))

    print("\n=== recovery: clean compositions of known units (HARD) ===")
    compositions = [
        (["ima", "motteru"], "今持ってる"),
        (["shudan", "no", "hani", "nara"], "手段の範囲なら"),
        (["ittan", "syuuryou", "demo", "iiyo"], "一旦終了でもいいよ"),
        (["ima", "motteru", "shudan", "no", "hani", "nara",
          "ittan", "syuuryou", "demo", "iiyo"],
         "今持ってる手段の範囲なら一旦終了でもいいよ"),
    ]
    clean_inputs = []
    for keys, expected in compositions:
        clean = "".join(keys)
        clean_inputs.append((clean, expected))
        res = general_phrase_rescue(clean, index)
        out = res[0] if res else None
        check(f"clean recover: {clean}", out == expected, f"got={out!r} want={expected!r}")

    print("\n=== recovery: reading-preserving corruptions (HARD) ===")
    safe_ops = [("style_flip", op_style_flip), ("sokuon_ime", op_sokuon_ime)]
    for clean, expected in clean_inputs:
        for opname, op in safe_ops:
            corrupt = op(clean)
            if corrupt == clean:
                continue
            res = general_phrase_rescue(corrupt, index)
            out = res[0] if res else None
            check(f"{opname}: {corrupt}", out == expected, f"got={out!r} want={expected!r}")

    print("\n=== recovery: reading-altering corruptions (REPORT ONLY) ===")
    risky_ops = [("run_triple", op_run), ("drop_geminate", op_drop_geminate),
                 ("long_vowel", op_long_vowel_inflate)]
    for clean, expected in clean_inputs:
        for opname, op in risky_ops:
            corrupt = op(clean)
            if corrupt == clean:
                continue
            res = general_phrase_rescue(corrupt, index)
            report(f"{opname}: {corrupt}", res[0] if res else None)

    print("\n=== negative gate: heavy / ambiguous noise must abstain (HARD) ===")
    negatives = [
        "xqzkwbvfjpmnlrtxqz",
        "zzzzzzzzzzzzzzzz",
        "qwlkjghfdspqwlkjghfds",
    ]
    for neg in negatives:
        res = general_phrase_rescue(neg, index)
        check(f"abstain: {neg}", res is None, f"got={res[0] if res else None!r}")

    print("\n=== contamination guard: target intent recovers when typed closer (HARD) ===")
    # The target sentence IS recoverable once the *severe* corruptions are typed
    # closer to canonical: moxtuteru (IME small-tsu) -> 持ってる, haninara -> 範囲なら,
    # ittan -> 一旦. Proves the route is general, not a single-string fit.
    closer = "imamoxtuterusyudannohaninaraittansyuuryoudemoiiyo"
    res = general_phrase_rescue(closer, index)
    out = res[0] if res else None
    want = "今持ってる手段の範囲なら一旦終了でもいいよ"
    check("closer-typing recovers full phrase", out == want, f"got={out!r}")

    print("\n=== recovery: mixed case / separators with reusable noise (HARD) ===")
    noisy = "IMAMO-XTUteWRUU SYUDANNO-HANINARA ITANNSYYYUURYOUDEMOIIYO"
    res = general_phrase_rescue(normalize_input(noisy), index)
    out = res[0] if res else None
    check("case/space/hyphen + tewruu + itannsyyyuu recovers", out == want, f"got={out!r}")

    print("\n=== safety guard: real target case must abstain (HARD) ===")
    target = "imamoxtutewruusyudannowataraitannsyyyuuryoudemoiiyo"
    res = general_phrase_rescue(target, index)
    check("target with watara ambiguity abstains", res is None, f"got={res[0] if res else None!r}")
    if res:
        m = res[1]
        report("target meta",
               f"anchor={m['anchor_ratio']:.2f} fill={m['fill_ratio']:.2f} "
               f"drops={m['drops']} cost/char={m['cost_per_char']:.3f}")

    print(f"\n==== {len(PASS)} passed, {len(FAIL)} failed ====")
    if FAIL:
        print("FAILURES:")
        for f in FAIL:
            print("  -", f)
        raise SystemExit(1)


if __name__ == "__main__":
    main()