Sam-Just / patterns.py
stevafernandes's picture
Upload 13 files
2bf83b3 verified
Raw
History Blame Contribute Delete
13.8 kB
"""Measured patterns across recordings (no AI).
* A comparison table of headline metrics.
* Pairwise acoustic similarity from three components that work even with two
files (no cross-file standardisation): timbre (MFCC distance scaled by the
recordings' own variability), spectral balance (Hellinger affinity of band
energies) and voice pitch (overlap of pitch histograms).
* Pairwise content similarity (TF-IDF cosine of transcripts), shared and
distinctive vocabulary.
* With 3+ files: closest pair / most distinct file and monotonic trends over
recording dates; with 4+: robust outliers and similarity groups.
"""
from __future__ import annotations
import itertools
import re
import numpy as np
METRICS = [
# key, label, unit, decimals, higher-is (for wording only)
("duration_min", "Duration", "min", 1),
("lufs", "Loudness", "LUFS", 1),
("snr_db", "Signal-to-noise (est.)", "dB", 0),
("activity_pct", "Sound activity", "%", 0),
("pauses_per_min", "Pauses (0.5 s+) per minute", "/min", 1),
("longest_pause_s", "Longest pause", "s", 1),
("pitch_median_hz", "Median voice pitch", "Hz", 0),
("centroid_hz", "Spectral centroid", "Hz", 0),
("bandwidth_khz", "Effective bandwidth", "kHz", 1),
("words_per_minute", "Speaking rate", "words/min", 0),
("speakers", "Speakers", "", 0),
("dominant_share_pct", "Main speaker's talk share", "%", 0),
("turns_per_minute", "Speaker turns per minute", "/min", 1),
("questions", "Questions asked", "", 0),
("fillers_per_100", "Fillers per 100 words", "", 1),
("lexical_diversity", "Vocabulary variety (MATTR)", "", 2),
("sentiment_score", "Sentiment (AI, -1 to +1)", "", 2),
]
METRIC_INFO = {k: (label, unit, dec) for k, label, unit, dec in METRICS}
STOP_EXTRA = {"um", "uh", "erm", "er", "uhm", "umm", "hmm", "mm", "mhm", "ah", "eh", "yeah", "okay", "ok",
"like", "just", "really", "know", "think", "going", "gonna", "kind", "sort", "thing", "things",
"right", "lot", "actually", "mean", "maybe", "got", "get", "inaudible", "laughs", "laughter",
"don", "didn", "doesn", "isn", "wasn", "aren", "weren", "couldn", "wouldn", "shouldn", "won", "ain",
"say", "said", "says", "tell", "told", "come", "came", "did", "does", "having", "trying", "want",
"wanted", "sorry", "thank", "thanks", "guess", "yes", "sure", "pretty", "stuff", "feel", "feels",
"went", "goes", "make", "made", "way", "good", "great", "little", "bit", "able", "look", "need"}
def metric_values(rec: dict) -> dict:
ac, sm, an = rec.get("acoustics"), rec.get("speech") or {}, rec.get("analysis") or {}
v: dict = {}
if ac is not None:
v["duration_min"] = ac.duration_s / 60
v["lufs"] = ac.integrated_lufs
v["snr_db"] = None if ac.steady_signal else ac.snr_db
v["activity_pct"] = ac.activity_ratio * 100
v["pauses_per_min"] = ac.pauses_per_min
v["longest_pause_s"] = ac.longest_pause_s
v["pitch_median_hz"] = ac.pitch_median_hz
v["centroid_hz"] = ac.spectral_centroid_hz
v["bandwidth_khz"] = ac.bandwidth_hz / 1000
if sm:
v["words_per_minute"] = sm.get("words_per_minute")
v["speakers"] = sm.get("speakers")
v["dominant_share_pct"] = sm["dominant_speaker_share"] * 100 if sm.get("dominant_speaker_share") is not None else None
v["turns_per_minute"] = sm.get("turns_per_minute")
v["questions"] = sm.get("questions")
v["fillers_per_100"] = sm.get("fillers_per_100_words")
v["lexical_diversity"] = sm.get("lexical_diversity")
sent = an.get("sentiment") if isinstance(an, dict) else None
if sent and isinstance(sent.get("score"), (int, float)):
v["sentiment_score"] = max(-1.0, min(1.0, float(sent["score"])))
return {k: (float(x) if x is not None and np.isfinite(x) else None) for k, x in v.items()}
# --- acoustic similarity --------------------------------------------------------
def _timbre(a, b) -> float:
ma, mb = np.array(a.mfcc_mean[1:13]), np.array(b.mfcc_mean[1:13])
sa, sb = np.array(a.mfcc_std[1:13]), np.array(b.mfcc_std[1:13])
pooled = np.sqrt((sa ** 2 + sb ** 2) / 2 + 1e-9)
d = float(np.sqrt(np.mean(((ma - mb) / pooled) ** 2)))
return float(np.exp(-d))
def _spectral(a, b) -> float:
p = np.array(list(a.band_fractions.values()))
q = np.array(list(b.band_fractions.values()))
bc = float(np.sum(np.sqrt(np.clip(p, 0, None) * np.clip(q, 0, None))))
return float(1 - np.sqrt(max(0.0, 1 - bc)))
def _pitch(a, b) -> float | None:
if not a.pitch_hist or not b.pitch_hist:
return None
return float(np.minimum(np.array(a.pitch_hist), np.array(b.pitch_hist)).sum())
def acoustic_similarity(recs: list[dict]) -> dict:
n = len(recs)
comps = {"timbre": np.full((n, n), np.nan), "spectral balance": np.full((n, n), np.nan),
"voice pitch": np.full((n, n), np.nan)}
for i, j in itertools.product(range(n), range(n)):
a, b = recs[i]["acoustics"], recs[j]["acoustics"]
if i == j:
for m in comps.values():
m[i, j] = 1.0
continue
if j < i:
continue
vals = {"timbre": _timbre(a, b), "spectral balance": _spectral(a, b), "voice pitch": _pitch(a, b)}
for k, v in vals.items():
if v is not None:
comps[k][i, j] = comps[k][j, i] = v
stack = np.stack(list(comps.values()))
with np.errstate(all="ignore"):
overall = np.nanmean(stack, axis=0)
return {"overall": overall, "components": comps}
# --- content similarity -----------------------------------------------------------
def content_similarity(texts: list[str]) -> dict | None:
"""TF-IDF cosine similarity plus shared / distinctive terms (2+ transcripts)."""
if sum(1 for t in texts if len(t.split()) >= 30) < 2:
return None
from sklearn.feature_extraction.text import ENGLISH_STOP_WORDS, CountVectorizer, TfidfTransformer
stop = sorted(set(ENGLISH_STOP_WORDS) | STOP_EXTRA)
cleaned = [re.sub(r"\[[^\]]*\]", " ", t) for t in texts]
cv = CountVectorizer(stop_words=stop, token_pattern=r"(?u)\b[^\W\d_]{3,}\b", ngram_range=(1, 2),
max_features=30000, lowercase=True)
try:
counts = cv.fit_transform(cleaned)
except ValueError: # empty vocabulary
return None
terms = np.array(cv.get_feature_names_out())
tfidf = TfidfTransformer(sublinear_tf=True).fit_transform(counts)
sim = (tfidf @ tfidf.T).toarray()
dense = counts.toarray()
df = (dense > 0).sum(axis=0)
n = len(texts)
shared = []
if n >= 2:
mask = df >= 2
order = np.argsort(-(dense[:, mask].sum(axis=0)))
idx = np.flatnonzero(mask)[order][:25]
shared = [{"term": terms[i], "files": int(df[i]), "count": int(dense[:, i].sum())} for i in idx]
distinctive = []
tf_dense = tfidf.toarray()
for d in range(n):
own = np.flatnonzero((dense[d] >= 2) & (df == 1))
pool = own if own.size >= 5 else np.flatnonzero(dense[d] >= 2)
if pool.size == 0:
pool = np.flatnonzero(dense[d] > 0)
top = pool[np.argsort(-tf_dense[d, pool])][:12]
distinctive.append([terms[i] for i in top])
return {"matrix": sim, "shared_terms": shared, "distinctive_terms": distinctive}
# --- trends / outliers / groups ------------------------------------------------------
def trends(recs: list[dict], values: list[dict]) -> list[dict]:
dated = [(r["info"].recorded_at, v, r["id"]) for r, v in zip(recs, values) if r["info"].recorded_at]
if len(dated) < 3:
return []
dated.sort(key=lambda x: x[0])
out = []
for key, label, unit, dec in METRICS:
series = [(d, v.get(key), fid) for d, v, fid in dated if v.get(key) is not None]
if len(series) < 3:
continue
ys = np.array([s[1] for s in series])
if np.ptp(ys) == 0:
continue
diffs = np.diff(ys)
if np.all(diffs > 0) or np.all(diffs < 0):
from scipy.stats import spearmanr
rho = float(spearmanr(np.arange(len(ys)), ys).statistic)
out.append({
"metric": key, "label": label, "unit": unit,
"direction": "increases" if diffs[0] > 0 else "decreases",
"rho": round(rho, 2), "n": len(ys),
"first": round(float(ys[0]), dec), "last": round(float(ys[-1]), dec),
"file_ids": [s[2] for s in series],
})
return out
def outliers(recs: list[dict], values: list[dict]) -> list[dict]:
if len(recs) < 4:
return []
out = []
for key, label, unit, dec in METRICS:
pairs = [(r["id"], v.get(key)) for r, v in zip(recs, values) if v.get(key) is not None]
if len(pairs) < 4:
continue
xs = np.array([p[1] for p in pairs])
med = float(np.median(xs))
mad = float(np.median(np.abs(xs - med)))
if mad <= 1e-9:
continue
for fid, x in pairs:
z = 0.6745 * (x - med) / mad
if abs(z) > 3.5:
out.append({"file_id": fid, "metric": key, "label": label, "unit": unit,
"value": round(x, dec), "median": round(med, dec), "z": round(float(z), 1)})
return out
def groups(ids: list[str], sim: np.ndarray) -> list[list[str]]:
n = len(ids)
if n < 4 or not np.isfinite(sim).all():
return []
from scipy.cluster.hierarchy import fcluster, linkage
from scipy.spatial.distance import squareform
dist = np.clip(1 - sim, 0, None)
np.fill_diagonal(dist, 0)
z = linkage(squareform(dist, checks=False), method="average")
heights = z[:, 2]
gaps = np.diff(heights)
if gaps.size == 0 or gaps.max() < 0.05:
return []
cut = heights[int(np.argmax(gaps))] + gaps.max() / 2
labels = fcluster(z, t=cut, criterion="distance")
out = [[ids[i] for i in range(n) if labels[i] == g] for g in sorted(set(labels))]
if len(out) < 2 or len(out) == n:
return []
return sorted(out, key=len, reverse=True)
# --- main ---------------------------------------------------------------------------------
def compare(recs: list[dict]) -> dict:
"""recs: successful files [{id, info, acoustics, speech, analysis, listening}]."""
ids = [r["id"] for r in recs]
values = [metric_values(r) for r in recs]
table = []
for key, label, unit, dec in METRICS:
row = {fid: v.get(key) for fid, v in zip(ids, values)}
if any(x is not None for x in row.values()):
table.append({"key": key, "label": label, "unit": unit, "decimals": dec, "values": row})
result: dict = {"ids": ids, "table": table, "values": values}
if len(recs) < 2:
return result
ac = acoustic_similarity(recs)
result["acoustic"] = ac
texts = [r["listening"].transcript_text() if r.get("listening") else "" for r in recs]
content = content_similarity([re.sub(r"^\[[^\]]*\] [^:]*: ", "", t, flags=re.M) for t in texts])
result["content"] = content
combined = ac["overall"].copy()
if content is not None:
with np.errstate(all="ignore"):
combined = np.nanmean(np.stack([ac["overall"], content["matrix"]]), axis=0)
result["combined"] = combined
n = len(ids)
pairs = [(combined[i, j], ids[i], ids[j]) for i in range(n) for j in range(i + 1, n) if np.isfinite(combined[i, j])]
if n >= 3 and pairs:
best = max(pairs)
result["closest_pair"] = {"ids": [best[1], best[2]], "similarity": round(float(best[0]), 3)}
with np.errstate(all="ignore"):
mean_sim = [(np.nanmean([combined[i, j] for j in range(n) if j != i]), ids[i]) for i in range(n)]
low = min(mean_sim)
result["most_distinct"] = {"id": low[1], "mean_similarity": round(float(low[0]), 3)}
result["trends"] = trends(recs, values)
result["outliers"] = outliers(recs, values)
result["groups"] = groups(ids, combined)
return result
def brief(cmp: dict) -> str:
"""Plain-text version for the synthesis prompt."""
ids = cmp["ids"]
lines = ["metric | " + " | ".join(ids)]
for row in cmp["table"]:
vals = []
for fid in ids:
x = row["values"].get(fid)
vals.append("n/a" if x is None else f"{x:.{row['decimals']}f}")
unit = f" ({row['unit']})" if row["unit"] else ""
lines.append(f"{row['label']}{unit} | " + " | ".join(vals))
if "acoustic" in cmp:
n = len(ids)
sims = []
for i in range(n):
for j in range(i + 1, n):
a = cmp["acoustic"]["overall"][i, j]
c = cmp["content"]["matrix"][i, j] if cmp.get("content") else None
s = f"{ids[i]}-{ids[j]}: acoustic similarity {a:.2f}"
if c is not None:
s += f", transcript vocabulary similarity {c:.2f}"
sims.append(s)
lines.append("Pairwise similarity (0-1): " + "; ".join(sims[:60]))
if cmp.get("content") and cmp["content"]["shared_terms"]:
lines.append("Terms shared by several transcripts: " + ", ".join(t["term"] for t in cmp["content"]["shared_terms"][:20]))
for t in cmp.get("trends", []):
lines.append(f"Trend over recording dates: {t['label']} {t['direction']} ({t['first']} -> {t['last']}, n={t['n']})")
for o in cmp.get("outliers", []):
lines.append(f"Outlier: {o['file_id']} {o['label']} = {o['value']} vs median {o['median']}")
if cmp.get("groups"):
lines.append("Similarity groups: " + " | ".join(", ".join(g) for g in cmp["groups"]))
return "\n".join(lines)