| |
| """Audit 2: segmentation, capitalisation, ponctuation finale, confusions de caracteres. |
| Toutes les estimations sont converties en points de SCORE (1 - 0.5WER - 0.5CER).""" |
| import csv, json, re, sys |
| from collections import Counter, defaultdict |
|
|
| OUT = [] |
| def P(*a): |
| s = " ".join(str(x) for x in a) |
| OUT.append(s); print(s) |
|
|
| sub = {} |
| with open("/root/sub_LMA06.csv", encoding="utf-8") as f: |
| for row in csv.DictReader(f): |
| sub[row["ID"]] = row["Target"] |
| lang = json.load(open("/root/test_lang.json", encoding="utf-8")) |
| hyp = {"lin": [], "sna": []} |
| for k, v in sub.items(): |
| if lang.get(k) in hyp: hyp[lang[k]].append(v) |
|
|
| train, val = {"lin": [], "sna": []}, {"lin": [], "sna": []} |
| for L in ("lin", "sna"): |
| for split, store in (("train", train), ("validation", val)): |
| with open("/scratch/prep/manifests/waxal_%s_%s.jsonl" % (L, split), encoding="utf-8") as f: |
| for line in f: |
| if line.strip(): store[L].append(json.loads(line)["text"]) |
|
|
| |
| NCLIP = 892 |
| tot_w = sum(len(t.split()) for L in hyp for t in hyp[L]) |
| tot_c = sum(len(t) for L in hyp for t in hyp[L]) |
| W_ERR = 0.5 / tot_w |
| C_ERR = 0.5 / tot_c |
| P("=== ECHELLE ===") |
| P("mots hyp total=%d, chars hyp total=%d, clips=%d" % (tot_w, tot_c, NCLIP)) |
| P("1 erreur-MOT = %.3e points de score" % W_ERR) |
| P("1 erreur-CARACTERE= %.3e points de score" % C_ERR) |
| P("ECART AU 2e (0.000199) = %.1f erreurs-mot OU %.0f erreurs-caractere" % (0.000199/W_ERR, 0.000199/C_ERR)) |
|
|
| PUNCT = ".,;:!?\"()«»" |
| def strip_p(w): return w.strip(PUNCT) |
| def toks(t): return [x for x in (strip_p(w) for w in t.split()) if x] |
|
|
| |
| P("") |
| P("=== A. SEGMENTATION : opportunites JOIN / SPLIT ===") |
| for L in ("lin", "sna"): |
| uni = Counter(); bi = Counter() |
| for t in train[L]: |
| ws = [w.lower() for w in toks(t)] |
| uni.update(ws) |
| bi.update(zip(ws, ws[1:])) |
| uni_v = Counter(); bi_v = Counter() |
| for t in val[L]: |
| ws = [w.lower() for w in toks(t)] |
| uni_v.update(ws); bi_v.update(zip(ws, ws[1:])) |
|
|
| |
| join_rules = {} |
| for (a, b), cb in bi.items(): |
| j = a + b |
| cj = uni.get(j, 0) |
| if cj >= 10 and cj > 4 * max(cb, 1) and cb <= cj // 8: |
| join_rules[(a, b)] = (cj, cb) |
| |
| for j, cj in uni.items(): |
| if cj < 10: continue |
| for i in range(2, len(j) - 1): |
| a, b = j[:i], j[i:] |
| if uni.get(a, 0) >= 50 and uni.get(b, 0) >= 50 and bi.get((a, b), 0) == 0: |
| join_rules.setdefault((a, b), (cj, 0)) |
|
|
| hits = Counter(); nhit = 0 |
| for t in hyp[L]: |
| ws = [w.lower() for w in toks(t)] |
| for pair in zip(ws, ws[1:]): |
| if pair in join_rules: |
| hits[pair] += 1; nhit += 1 |
| |
| tp = fp = 0 |
| for (a, b), (cj, cb) in join_rules.items(): |
| tp += uni_v.get(a + b, 0); fp += bi_v.get((a, b), 0) |
| prec = tp / max(tp + fp, 1) |
| P("%s | JOIN: %d regles, %d declenchements dans hyp (%.2f%% des mots)" % (L, len(join_rules), nhit, 100.0*nhit/sum(len(toks(t)) for t in hyp[L]))) |
| P("%s | JOIN precision held-out VAL: %d formes jointes vs %d formes separees -> %.1f%%" % (L, tp, fp, 100*prec)) |
| P("%s | top JOIN declenches: %s" % (L, ", ".join("%s+%s x%d[tr %d/%d]" % (a, b, c, join_rules[(a,b)][0], join_rules[(a,b)][1]) for (a, b), c in hits.most_common(15)))) |
| gain_w = nhit * (2*prec - 1) * 2 |
| P("%s | GAIN JOIN estime = %d fix x (2p-1)=%.2f x 2 err = %.1f err-mot = %+.5f score" % (L, nhit, 2*prec-1, gain_w, gain_w*W_ERR)) |
|
|
| |
| split_hits = Counter(); nsplit = 0 |
| for t in hyp[L]: |
| for w in toks(t): |
| wl = w.lower() |
| if uni.get(wl, 0) > 0 or len(wl) < 5: continue |
| best = None |
| for i in range(2, len(wl) - 1): |
| a, b = wl[:i], wl[i:] |
| cb = bi.get((a, b), 0) |
| if cb >= 5 and (best is None or cb > best[1]): best = ((a, b), cb) |
| if best: |
| split_hits[best[0]] += 1; nsplit += 1 |
| P("%s | SPLIT: %d mots OOV decomposables en bigram frequent du train (%.2f%% des mots)" % (L, nsplit, 100.0*nsplit/sum(len(toks(t)) for t in hyp[L]))) |
| P("%s | top SPLIT: %s" % (L, ", ".join("%s|%s x%d" % (a, b, c) for (a, b), c in split_hits.most_common(12)))) |
| P("%s | GAIN SPLIT max (p=1) = %.1f err-mot = %+.5f score" % (L, nsplit*2.0, nsplit*2.0*W_ERR)) |
|
|
| |
| P("") |
| P("=== B. CAPITALISATION : noms propres perdus ===") |
| SENT_END = re.compile(r"[.!?]$") |
| def positions(t): |
| """rend (mot, est_debut_de_phrase)""" |
| ws = t.split(); res = []; start = True |
| for w in ws: |
| res.append((w, start)) |
| start = bool(SENT_END.search(w)) |
| return res |
| for L in ("lin", "sna"): |
| cap = Counter(); low = Counter() |
| for t in train[L] + val[L]: |
| for w, st in positions(t): |
| c = strip_p(w) |
| if not c or not c[0].isalpha() or st: continue |
| (cap if c[0].isupper() else low)[c.lower()] += 1 |
| |
| lex = {w: (cap[w], low.get(w, 0)) for w in cap if cap[w] >= 3 and cap[w] >= 4*low.get(w, 0)} |
| P("%s | lexique noms propres (cap>=3 et cap>=4x low, hors debut de phrase): %d entrees" % (L, len(lex))) |
| P("%s | exemples: %s" % (L, ", ".join("%s(%d/%d)" % (w, c, l) for w, (c, l) in sorted(lex.items(), key=lambda x: -x[1][0])[:20]))) |
| miss = Counter(); nmiss = 0; ok = 0 |
| for t in hyp[L]: |
| for w, st in positions(t): |
| c = strip_p(w) |
| if not c or st: continue |
| if c.lower() in lex: |
| if c[0].isupper(): ok += 1 |
| else: miss[c.lower()] += 1; nmiss += 1 |
| P("%s | dans HYP: %d occurrences correctement capitalisees, %d MANQUEES" % (L, ok, nmiss)) |
| P("%s | top manquees: %s" % (L, ", ".join("%s x%d" % (w, c) for w, c in miss.most_common(20)))) |
| |
| tot_lex = sum(c + l for c, l in lex.values()); tot_cap = sum(c for c, l in lex.values()) |
| p = tot_cap / max(tot_lex, 1) |
| P("%s | precision attendue de la regle = %.1f%%" % (L, 100*p)) |
| g = nmiss*(2*p-1) |
| P("%s | GAIN CAPITALISATION = %.1f err-mot + %.1f err-car = %+.5f score" % (L, g, g, g*W_ERR + g*C_ERR)) |
|
|
| |
| P("") |
| P("=== C. POLITIQUE DU POINT FINAL (1 decision / clip) ===") |
| for L in ("lin", "sna"): |
| ref_rate = sum(1 for t in train[L]+val[L] if t.strip().endswith(".")) / len(train[L]+val[L]) |
| hyp_rate = sum(1 for t in hyp[L] if t.strip().endswith(".")) / len(hyp[L]) |
| n = len(hyp[L]) |
| P("%s | P(ref se termine par '.') = %.4f ; notre taux = %.4f (n=%d clips)" % (L, ref_rate, hyp_rate, n)) |
| |
| def cost(q): return q*(1-ref_rate) + (1-q)*ref_rate |
| c_now, c_all, c_none = cost(hyp_rate), cost(1.0), cost(0.0) |
| P("%s | taux d'erreur point-final: actuel=%.4f toujours-'.'=%.4f jamais-'.'=%.4f" % (L, c_now, c_all, c_none)) |
| best = min((c_all, "TOUJOURS"), (c_none, "JAMAIS"), (c_now, "ACTUEL")) |
| d = (c_now - c_all) * n |
| P("%s | politique constante optimale = %s ; passer a TOUJOURS: %+.1f clips corriges = %+.5f score (borne haute, suppose independance)" % (L, best[1], d, d*(W_ERR + C_ERR))) |
|
|
| |
| P("") |
| P("=== D. DEFICIT DE VIRGULES : cout theorique (rappel: insertion deja refutee au LB) ===") |
| for L in ("lin", "sna"): |
| tw = sum(len(t.split()) for t in train[L]+val[L]) |
| rate = sum(t.count(",") for t in train[L]+val[L]) / tw |
| hw = sum(len(t.split()) for t in hyp[L]) |
| hrate = sum(t.count(",") for t in hyp[L]) / hw |
| manque = (rate - hrate) * hw |
| P("%s | virgules/mot: ref=%.4f hyp=%.4f -> %.0f virgules manquantes sur le test" % (L, rate, hrate, manque)) |
| P("%s | cout actuel = %.0f err-mot + %.0f err-car = %.5f score (p>50%% requis pour gagner)" % (L, manque, manque, manque*W_ERR + manque*C_ERR)) |
|
|
| |
| P("") |
| P("=== E. CONFUSIONS DE CARACTERES (mots OOV a distance 1 d'un mot du train) ===") |
| def ed1_ops(a, b): |
| """retourne l'operation unique transformant a en b, ou None""" |
| if a == b: return None |
| la, lb = len(a), len(b) |
| if abs(la - lb) > 1: return None |
| if la == lb: |
| d = [i for i in range(la) if a[i] != b[i]] |
| if len(d) == 1: return ("SUB", a[d[0]], b[d[0]]) |
| return None |
| if la > lb: a, b, kind = b, a, "DEL" |
| else: kind = "INS" |
| i = 0 |
| while i < len(a) and a[i] == b[i]: i += 1 |
| if a[i:] == b[i+1:]: |
| return ("SUP" if kind == "DEL" else "MANQUE", b[i], "") |
| return None |
| for L in ("lin", "sna"): |
| uni = Counter() |
| for t in train[L]+val[L]: uni.update(w.lower() for w in toks(t)) |
| bylen = defaultdict(list) |
| for w in uni: bylen[len(w)].append(w) |
| ops = Counter(); n_res = 0; n_oov = 0 |
| examples = defaultdict(list) |
| for t in hyp[L]: |
| for w in toks(t): |
| wl = w.lower() |
| if wl in uni or len(wl) < 4: continue |
| n_oov += 1 |
| cands = [] |
| for cl in (len(wl)-1, len(wl), len(wl)+1): |
| for c in bylen.get(cl, []): |
| if c[0] != wl[0] and c[-1] != wl[-1]: continue |
| op = ed1_ops(wl, c) |
| if op: cands.append((uni[c], op, c)) |
| if cands: |
| cands.sort(reverse=True) |
| _, op, c = cands[0] |
| ops[op] += 1; n_res += 1 |
| if len(examples[op]) < 3: examples[op].append("%s->%s" % (wl, c)) |
| P("%s | %d mots OOV analyses, %d a distance 1 d'un mot du train (%.0f%%)" % (L, n_oov, n_res, 100.0*n_res/max(n_oov,1))) |
| for op, c in ops.most_common(18): |
| kind = op[0] |
| lbl = ("%s -> %s" % (op[1], op[2])) if kind == "SUB" else ("caractere en trop: %r" % op[1] if kind == "SUP" else "caractere manquant: %r" % op[1]) |
| P(" %-6s %-28s x%-4d ex: %s" % (kind, lbl, c, ", ".join(examples[op]))) |
|
|
| with open("/root/audit_STATSLING2_out.txt", "w", encoding="utf-8") as f: |
| f.write("\n".join(OUT)) |
| print("\n[OK] /root/audit_STATSLING2_out.txt") |
|
|