romaji2ja / code /eval_general_phrase.py
limoXD's picture
Publish accepted A75 checkpoint with bound evaluation evidence
03b56f8 verified
Raw
History Blame Contribute Delete
8.65 kB
"""Self-checking evaluation for the generic noisy-romaji rescue.
This is deliberately *generative*: recovery cases are produced by applying
corruption operators to clean compositions of known dictionary/colloquial
units, so the suite proves the mechanism generalizes rather than memorizing one
string. It also asserts the confidence gate *abstains* on heavy/ambiguous noise.
Corruptions are split by whether they preserve the intended *reading*:
* reading-preserving (style flip wapuro<->Hepburn, IME small-tsu spelling) are
hard-asserted to recover exactly -- these are the realistic, high-value wins;
* reading-altering (triardupling, geminate drop, vowel inflation) are reported
only, because a corrupted form may legitimately collide with another real word
(e.g. ``itan`` -> 異端 vs ``ittan`` -> 一旦) and forcing one answer would be the
very overfitting we are avoiding.
Run:
python src/eval_general_phrase.py
Exits non-zero if any hard assertion fails.
"""
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from normalization import normalize_input
from romaji_kana import load_general_lexicon
from general_phrase import (
build_general_phrase_index,
canon_style,
canonicalize_romaji_variants,
general_phrase_rescue,
_collapse_runs,
_geminate_sokuon,
)
DEFAULT_GENERAL_LEXICON = "artifacts/lexicon/general_reading_lexicon.json"
PASS = []
FAIL = []
def check(name, cond, detail=""):
(PASS if cond else FAIL).append(name)
mark = "ok " if cond else "FAIL"
line = f"[{mark}] {name}"
if detail:
line += f" :: {detail}"
print(line)
def report(name, value):
print(f"[rpt ] {name} :: {value}")
# --------------------------------------------------------------------------- #
# Corruption operators (deterministic; no RNG)
# --------------------------------------------------------------------------- #
def op_style_flip(s):
# wapuro -> Hepburn-ish flips that canon_style must absorb (reading-preserving)
for a, b in (("sy", "sh"), ("ty", "ch"), ("tu", "tsu"), ("hu", "fu"), ("zy", "j")):
if a in s:
return s.replace(a, b, 1)
return s
def op_sokuon_ime(s):
# a geminate consonant -> IME small-tsu spelling (reading-preserving)
for i in range(len(s) - 1):
if s[i] == s[i + 1] and s[i] in "bcdfghjkmpqrstvwyz":
return s[:i] + "xtu" + s[i + 1:]
return s
def op_run(s):
# triple the first consonant/glide (may alter reading near geminates)
for i, ch in enumerate(s):
if ch in "ybcdfghjkmpqrstvwz":
return s[:i] + ch * 3 + s[i + 1:]
return s
def op_drop_geminate(s):
for i in range(len(s) - 1):
if s[i] == s[i + 1] and s[i] in "bcdfghjkmpqrstvwyz":
return s[:i] + s[i + 1:]
return s
def op_long_vowel_inflate(s):
for i, ch in enumerate(s):
if ch in "aiueo":
return s[:i] + ch + ch + s[i:]
return s
# --------------------------------------------------------------------------- #
def main():
g = load_general_lexicon(DEFAULT_GENERAL_LEXICON)
index = build_general_phrase_index(g)
print("=== canonicalizer unit tests ===")
check("canon sh->sy", canon_style("shudan") == "syudan", canon_style("shudan"))
check("canon sy stable", canon_style("syudan") == "syudan")
check("canon shuuryou->syuuryou", canon_style("shuuryou") == "syuuryou", canon_style("shuuryou"))
check("canon tsu->tu", canon_style("tsunami") == "tunami", canon_style("tsunami"))
check("canon chi->ti", canon_style("chizu") == "tizu", canon_style("chizu"))
check("canon fu->hu", canon_style("fujisan") == "huzisan", canon_style("fujisan"))
check("canon ji->zi", canon_style("jisho") == "zisyo", canon_style("jisho"))
check("sokuon moxtute->motte", _geminate_sokuon("moxtute") == "motte", _geminate_sokuon("moxtute"))
check("sokuon ixtukai->ikkai", _geminate_sokuon("ixtukai") == "ikkai", _geminate_sokuon("ixtukai"))
check("sokuon kixtute->kitte", _geminate_sokuon("kixtute") == "kitte", _geminate_sokuon("kixtute"))
check("run syyyuu->syuu", _collapse_runs("syyyuu") == "syuu", _collapse_runs("syyyuu"))
check("run aaaa->aa (vowel keeps 2)", _collapse_runs("aaaa") == "aa", _collapse_runs("aaaa"))
check("run uu stable", _collapse_runs("uu") == "uu")
variants = canonicalize_romaji_variants(
"imamoxtutewruusyudannowataraitannsyyyuuryoudemoiiyo"
)
check("variants fold motte+syuu",
any("motte" in v and "syuu" in v for v in variants),
" | ".join(variants))
check("variants repair tewruu only as additive candidate",
any("imamotteru" in v for v in variants),
" | ".join(variants))
print("\n=== recovery: clean compositions of known units (HARD) ===")
compositions = [
(["ima", "motteru"], "今持ってる"),
(["shudan", "no", "hani", "nara"], "手段の範囲なら"),
(["ittan", "syuuryou", "demo", "iiyo"], "一旦終了でもいいよ"),
(["ima", "motteru", "shudan", "no", "hani", "nara",
"ittan", "syuuryou", "demo", "iiyo"],
"今持ってる手段の範囲なら一旦終了でもいいよ"),
]
clean_inputs = []
for keys, expected in compositions:
clean = "".join(keys)
clean_inputs.append((clean, expected))
res = general_phrase_rescue(clean, index)
out = res[0] if res else None
check(f"clean recover: {clean}", out == expected, f"got={out!r} want={expected!r}")
print("\n=== recovery: reading-preserving corruptions (HARD) ===")
safe_ops = [("style_flip", op_style_flip), ("sokuon_ime", op_sokuon_ime)]
for clean, expected in clean_inputs:
for opname, op in safe_ops:
corrupt = op(clean)
if corrupt == clean:
continue
res = general_phrase_rescue(corrupt, index)
out = res[0] if res else None
check(f"{opname}: {corrupt}", out == expected, f"got={out!r} want={expected!r}")
print("\n=== recovery: reading-altering corruptions (REPORT ONLY) ===")
risky_ops = [("run_triple", op_run), ("drop_geminate", op_drop_geminate),
("long_vowel", op_long_vowel_inflate)]
for clean, expected in clean_inputs:
for opname, op in risky_ops:
corrupt = op(clean)
if corrupt == clean:
continue
res = general_phrase_rescue(corrupt, index)
report(f"{opname}: {corrupt}", res[0] if res else None)
print("\n=== negative gate: heavy / ambiguous noise must abstain (HARD) ===")
negatives = [
"xqzkwbvfjpmnlrtxqz",
"zzzzzzzzzzzzzzzz",
"qwlkjghfdspqwlkjghfds",
]
for neg in negatives:
res = general_phrase_rescue(neg, index)
check(f"abstain: {neg}", res is None, f"got={res[0] if res else None!r}")
print("\n=== contamination guard: target intent recovers when typed closer (HARD) ===")
# The target sentence IS recoverable once the *severe* corruptions are typed
# closer to canonical: moxtuteru (IME small-tsu) -> 持ってる, haninara -> 範囲なら,
# ittan -> 一旦. Proves the route is general, not a single-string fit.
closer = "imamoxtuterusyudannohaninaraittansyuuryoudemoiiyo"
res = general_phrase_rescue(closer, index)
out = res[0] if res else None
want = "今持ってる手段の範囲なら一旦終了でもいいよ"
check("closer-typing recovers full phrase", out == want, f"got={out!r}")
print("\n=== recovery: mixed case / separators with reusable noise (HARD) ===")
noisy = "IMAMO-XTUteWRUU SYUDANNO-HANINARA ITANNSYYYUURYOUDEMOIIYO"
res = general_phrase_rescue(normalize_input(noisy), index)
out = res[0] if res else None
check("case/space/hyphen + tewruu + itannsyyyuu recovers", out == want, f"got={out!r}")
print("\n=== safety guard: real target case must abstain (HARD) ===")
target = "imamoxtutewruusyudannowataraitannsyyyuuryoudemoiiyo"
res = general_phrase_rescue(target, index)
check("target with watara ambiguity abstains", res is None, f"got={res[0] if res else None!r}")
if res:
m = res[1]
report("target meta",
f"anchor={m['anchor_ratio']:.2f} fill={m['fill_ratio']:.2f} "
f"drops={m['drops']} cost/char={m['cost_per_char']:.3f}")
print(f"\n==== {len(PASS)} passed, {len(FAIL)} failed ====")
if FAIL:
print("FAILURES:")
for f in FAIL:
print(" -", f)
raise SystemExit(1)
if __name__ == "__main__":
main()