waxal2026-backup / archive /scripts /audit_STATSLING.py
Pricile's picture
Upload folder using huggingface_hub
0e11779 verified
Raw
History Blame Contribute Delete
11.8 kB
# -*- coding: utf-8 -*-
"""Audit linguistique comparatif: sorties sub_LMA06 vs corpus train, par langue."""
import csv, json, re, sys, unicodedata
from collections import Counter
OUT = []
def P(*a):
s = " ".join(str(x) for x in a)
OUT.append(s)
print(s)
SUB = "/root/sub_LMA06.csv"
LANG = "/root/test_lang.json"
MAN = "/scratch/prep/manifests/waxal_%s_%s.jsonl"
# ---------- chargement ----------
sub = {}
with open(SUB, encoding="utf-8") as f:
r = csv.DictReader(f)
for row in r:
sub[row["ID"]] = row["Target"]
lang = json.load(open(LANG, encoding="utf-8"))
hyp = {"lin": [], "sna": []}
for k, v in sub.items():
L = lang.get(k)
if L in hyp:
hyp[L].append(v)
train = {"lin": [], "sna": []}
val = {"lin": [], "sna": []}
for L in ("lin", "sna"):
for split, store in (("train", train), ("validation", val)):
try:
with open(MAN % (L, split), encoding="utf-8") as f:
for line in f:
line = line.strip()
if line:
store[L].append(json.loads(line)["text"])
except FileNotFoundError:
pass
P("=== TAILLES ===")
for L in ("lin", "sna"):
P("%s: hyp=%d clips, train=%d phrases, val=%d phrases" % (L, len(hyp[L]), len(train[L]), len(val[L])))
# ---------- tokenisation ----------
WORD = re.compile(r"[^\W\d_]+", re.UNICODE) # lettres seulement, pour le vocabulaire
def words(t):
return WORD.findall(t.lower())
def words_cased(t):
return WORD.findall(t)
def toks_ws(t): # tokens WER (espaces), tels que le scoreur les voit
return t.split()
# ================= 1. OOV =================
P("")
P("=== 1. TAUX DE MOTS HORS-VOCABULAIRE (hyp vs vocab train, minuscules) ===")
oov_detail = {}
for L in ("lin", "sna"):
vtrain = Counter()
for t in train[L]:
vtrain.update(words(t))
vval = Counter()
for t in val[L]:
vval.update(words(t))
vall = Counter(vtrain); vall.update(vval)
hw = []
for t in hyp[L]:
hw.extend(words(t))
hc = Counter(hw)
n_tok = len(hw)
oov_tok_tr = sum(c for w, c in hc.items() if w not in vtrain)
oov_typ_tr = sum(1 for w in hc if w not in vtrain)
oov_tok_all = sum(c for w, c in hc.items() if w not in vall)
oov_typ_all = sum(1 for w in hc if w not in vall)
P("%s | tokens hyp=%d, types hyp=%d | vocab train=%d types (+val -> %d)" % (L, n_tok, len(hc), len(vtrain), len(vall)))
P("%s | OOV vs TRAIN : tokens %d (%.2f%%), types %d (%.2f%%)" % (L, oov_tok_tr, 100.0*oov_tok_tr/max(n_tok,1), oov_typ_tr, 100.0*oov_typ_tr/max(len(hc),1)))
P("%s | OOV vs TRAIN+VAL : tokens %d (%.2f%%), types %d (%.2f%%)" % (L, oov_tok_all, 100.0*oov_tok_all/max(n_tok,1), oov_typ_all, 100.0*oov_typ_all/max(len(hc),1)))
oov_words = sorted([(c, w) for w, c in hc.items() if w not in vall], reverse=True)[:40]
P("%s | top OOV (vs train+val): %s" % (L, ", ".join("%s(%d)" % (w, c) for c, w in oov_words)))
# OOV du train lui-meme mesure en leave-one-out approx: taux de mots du val absents du train
vw = []
for t in val[L]:
vw.extend(words(t))
vc = Counter(vw)
oov_val = sum(c for w, c in vc.items() if w not in vtrain)
P("%s | REFERENCE: OOV du VAL vs train = %.2f%% tokens (borne naturelle)" % (L, 100.0*oov_val/max(len(vw),1)))
oov_detail[L] = dict(n_tok=n_tok, oov_tok_all=oov_tok_all, rate=100.0*oov_tok_all/max(n_tok,1),
val_rate=100.0*oov_val/max(len(vw),1), vtrain=vtrain, vall=vall, hc=hc)
# ============ 2. LONGUEURS DE MOTS ============
P("")
P("=== 2. DISTRIBUTION DES LONGUEURS DE MOTS (caracteres) ===")
def lenstats(texts):
ws = []
for t in texts:
ws.extend(words(t))
if not ws:
return None
ls = [len(w) for w in ws]
ls.sort()
n = len(ls)
mean = sum(ls)/n
med = ls[n//2]
p90 = ls[int(0.9*n)]
return dict(n=n, mean=mean, med=med, p90=p90, dist=Counter(ls))
for L in ("lin", "sna"):
a = lenstats(hyp[L]); b = lenstats(train[L])
P("%s | HYP : n=%d moy=%.3f med=%d p90=%d" % (L, a["n"], a["mean"], a["med"], a["p90"]))
P("%s | TRAIN : n=%d moy=%.3f med=%d p90=%d" % (L, b["n"], b["mean"], b["med"], b["p90"]))
P("%s | ECART moy hyp-train = %+.3f car (%.2f%%)" % (L, a["mean"]-b["mean"], 100.0*(a["mean"]-b["mean"])/b["mean"]))
for src, d in (("HYP", a), ("TRAIN", b)):
tot = sum(d["dist"].values())
row = " ".join("%d:%.1f%%" % (k, 100.0*d["dist"].get(k,0)/tot) for k in range(1, 13))
P(" %s %s len>=13:%.1f%%" % (src, row, 100.0*sum(v for k,v in d["dist"].items() if k>=13)/tot))
# mots par clip / par seconde
wpc_h = a["n"]/len(hyp[L])
wpc_t = b["n"]/len(train[L])
P("%s | mots/clip HYP=%.2f TRAIN=%.2f (train clips potentiellement + longs)" % (L, wpc_h, wpc_t))
# ============ 3. TOP-40 MOTS FREQUENTS ============
P("")
P("=== 3. TOP-40 MOTS DU TRAIN : frequence relative train vs hyp ===")
for L in ("lin", "sna"):
tw = []
for t in train[L]:
tw.extend(words(t))
tc = Counter(tw); Nt = len(tw)
hc = oov_detail[L]["hc"]; Nh = oov_detail[L]["n_tok"]
P("--- %s ---" % L)
P(" %-16s %8s %8s %8s %9s" % ("mot", "f_train", "f_hyp", "ratio", "delta_tok"))
rows = []
for w, c in tc.most_common(40):
ft = 100.0*c/Nt
fh = 100.0*hc.get(w, 0)/Nh
ratio = fh/ft if ft > 0 else float("nan")
# nombre de tokens hyp en trop/manque si on alignait la frequence
delta = hc.get(w, 0) - ft/100.0*Nh
rows.append((w, ft, fh, ratio, delta))
for w, ft, fh, ratio, delta in rows:
flag = ""
if ratio < 0.72: flag = " <-- SOUS-PRODUIT"
elif ratio > 1.38: flag = " <-- SUR-PRODUIT"
P(" %-16s %7.3f%% %7.3f%% %6.2f %+8.1f%s" % (w, ft, fh, ratio, delta, flag))
# ============ 4. LIN vs SNA : agglutination ============
P("")
P("=== 4. LIN vs SNA : morphologie comparee (sur le TRAIN) ===")
for L in ("lin", "sna"):
tw = []
for t in train[L]:
tw.extend(words(t))
tc = Counter(tw); N = len(tw); V = len(tc)
hapax = sum(1 for w, c in tc.items() if c == 1)
dis = sum(1 for w, c in tc.items() if c == 2)
mean_len = sum(len(w) for w in tw)/N
# couverture: % de tokens couverts par les 1000 mots les plus frequents
top = tc.most_common(1000)
cov1k = 100.0*sum(c for _, c in top)/N
cov100 = 100.0*sum(c for _, c in tc.most_common(100))/N
# TTR normalise (sur 200k tokens echantillonnes deterministe)
sub_tw = tw[:200000]
ttr = 100.0*len(set(sub_tw))/max(len(sub_tw),1)
P("%s | tokens=%d types=%d TTR(200k)=%.2f%% moy_len=%.3f" % (L, N, V, ttr, mean_len))
P("%s | hapax=%d (%.2f%% des types, %.3f%% des tokens) dis-legomena=%d (%.2f%% types)" % (L, hapax, 100.0*hapax/V, 100.0*hapax/N, dis, 100.0*dis/V))
P("%s | couverture top-100=%.2f%% top-1000=%.2f%% des tokens" % (L, cov100, cov1k))
# caracteres par phrase / mots par phrase
nch = sum(len(t) for t in train[L]); nph = len(train[L])
P("%s | phrase moy: %.1f car, %.2f mots" % (L, nch/nph, N/nph))
# ============ 5. CARACTERES / PONCTUATION / CASSE ============
P("")
P("=== 5. INVENTAIRE CARACTERES, PONCTUATION, CASSE ===")
for L in ("lin", "sna"):
ch_t = Counter(); ch_h = Counter()
for t in train[L]: ch_t.update(t)
for t in hyp[L]: ch_h.update(t)
Nt = sum(ch_t.values()); Nh = sum(ch_h.values())
P("--- %s --- chars train=%d hyp=%d" % (L, Nt, Nh))
keys = set(ch_t) | set(ch_h)
punct = sorted([c for c in keys if not c.isalnum() and not c.isspace()])
P(" ponctuation/symboles (train ‰ | hyp ‰ | ratio):")
for c in punct:
a = 1000.0*ch_t.get(c,0)/Nt; b = 1000.0*ch_h.get(c,0)/Nh
r = (b/a) if a > 0 else float("inf")
name = unicodedata.name(c, "?")
P(" %-3s U+%04X %-28s %7.3f | %7.3f | %s" % (repr(c)[1:-1], ord(c), name[:28], a, b, ("%.2f" % r) if a>0 else "INF"))
# lettres presentes dans train mais jamais/rarement en hyp
alpha_t = {c: n for c, n in ch_t.items() if c.isalpha()}
alpha_h = {c: n for c, n in ch_h.items() if c.isalpha()}
miss = []
for c, n in sorted(alpha_t.items(), key=lambda x: -x[1]):
a = 1000.0*n/Nt; b = 1000.0*alpha_h.get(c,0)/Nh
if a >= 0.02 and (b == 0 or b/a < 0.5 or b/a > 2.0):
miss.append((c, a, b))
P(" lettres desequilibrees (>=0.02 pour-mille dans train, ratio hors [0.5,2]):")
for c, a, b in miss[:30]:
P(" %-3s U+%04X %7.3f | %7.3f | %s" % (repr(c)[1:-1], ord(c), a, b, ("%.2f" % (b/a)) if a>0 else "-"))
extra = [(c, 1000.0*n/Nh) for c, n in alpha_h.items() if c not in alpha_t]
P(" lettres en HYP absentes du TRAIN: %s" % (", ".join("%s(%.3f‰)" % (c, v) for c, v in sorted(extra, key=lambda x:-x[1])) or "aucune"))
# casse
def case_stats(texts):
n = len(texts); up1 = 0; endpt = Counter(); capmid = 0; wtot = 0; allcaps = 0
for t in texts:
t2 = t.strip()
if not t2: continue
if t2[0].isupper(): up1 += 1
endpt[t2[-1]] += 1
ws = t2.split()
for i, w in enumerate(ws):
wtot += 1
core = WORD.findall(w)
if not core: continue
c0 = core[0]
if i > 0 and c0[0].isupper(): capmid += 1
if len(c0) > 1 and c0.isupper(): allcaps += 1
return n, up1, endpt, capmid, wtot, allcaps
for src, texts in (("TRAIN", train[L]), ("HYP", hyp[L])):
n, up1, endpt, capmid, wtot, allcaps = case_stats(texts)
P(" %s casse: 1re lettre majuscule %.2f%% | mots capitalises hors-initiale %.2f%% | ALLCAPS %.2f%%" % (src, 100.0*up1/n, 100.0*capmid/max(wtot,1), 100.0*allcaps/max(wtot,1)))
top_end = ", ".join("%r:%.1f%%" % (c, 100.0*v/n) for c, v in endpt.most_common(6))
P(" %s dernier caractere: %s" % (src, top_end))
# ponctuation par mot
for src, texts in (("TRAIN", train[L]), ("HYP", hyp[L])):
nw = sum(len(t.split()) for t in texts)
ncomma = sum(t.count(",") for t in texts)
nper = sum(t.count(".") for t in texts)
P(" %s: virgules/100mots=%.2f points/100mots=%.2f" % (src, 100.0*ncomma/nw, 100.0*nper/nw))
# ============ 6. n' (Shona velar nasal) et digrammes ============
P("")
P("=== 6. MOTIFS ORTHOGRAPHIQUES CIBLES ===")
for L in ("lin", "sna"):
def pat(texts, rx):
c = 0
for t in texts:
c += len(re.findall(rx, t))
return c
ntw = sum(len(t.split()) for t in train[L])
nhw = sum(len(t.split()) for t in hyp[L])
for label, rx in (("n' (nasal velaire)", r"n['’]"), ("apostrophe droite '", r"'"), ("apostrophe typo ’", r"’"), ("trait d'union", r"-"), ("chiffres", r"\d")):
a = 1000.0*pat(train[L], rx)/ntw
b = 1000.0*pat(hyp[L], rx)/nhw
P("%s | %-22s train=%.3f/1000mots hyp=%.3f/1000mots ratio=%s" % (L, label, a, b, ("%.2f" % (b/a)) if a > 0 else "n/a"))
# ============ 7. longueur des phrases hyp vs attendu ============
P("")
P("=== 7. LONGUEUR DES SORTIES ===")
for L in ("lin", "sna"):
hl = sorted(len(t.split()) for t in hyp[L])
tl = sorted(len(t.split()) for t in train[L])
P("%s | HYP mots/phrase: moy=%.2f med=%d min=%d max=%d" % (L, sum(hl)/len(hl), hl[len(hl)//2], hl[0], hl[-1]))
P("%s | TRAIN mots/phrase: moy=%.2f med=%d min=%d max=%d" % (L, sum(tl)/len(tl), tl[len(tl)//2], tl[0], tl[-1]))
empt = sum(1 for t in hyp[L] if not t.strip())
P("%s | sorties vides: %d ; sorties <3 mots: %d" % (L, empt, sum(1 for x in hl if x < 3)))
with open("/root/audit_STATSLING_out.txt", "w", encoding="utf-8") as f:
f.write("\n".join(OUT))
print("\n[OK] ecrit /root/audit_STATSLING_out.txt")