#!/usr/bin/env python3 """Render an honest English study log; pending experiments have no invented scores.""" import html import json from datetime import datetime, timezone from pathlib import Path from study_paths import ROOT, CODE, BENCH, GEMINI, RELEASE, read OUT = CODE.parent / 'WHISPER_EMBEDDING_PROBE_COMPARISON_2026-10-06.html' def esc(value): return html.escape(str(value)) def number(value): return 'Pending / unavailable' if value is None else f'{value:.4f}' def optional(path): return read(path) if path.exists() else {} def table(headers, rows): return '
| '+esc(x)+' | ' for x in headers) + '
|---|
| '+esc(x)+' | ' for x in row)+'
A controlled downstream-head study of emotion, voice style, quality, speaker embeddings and vocal-burst localization. This page records actual execution state and fills result tables as completed artifacts become available.
', 'Updated '+esc(datetime.now(timezone.utc).isoformat())+'. Existing public benchmark report · Earlier supervised score adapters · Released Whisper weights.
', 'These bounded checks verify software compatibility and finite training updates. They are not scientific evaluation results. A stale console launcher used the system Python instead of the active production interpreter; affected queued scripts were replaced with python -m torch.distributed.run. The live job IDs below are the replacements. Final Flash targets remain a required gate; unresolved or invalid annotations are retained for review and excluded from fine-tuning.
Base and Small each have a dedicated four-GPU full fine-tuning allocation for two epochs. Every encoder weight, including the initially fixed positional embedding table, and every multitask head is trainable. The jobs are pre-submitted on hold and released immediately when the final Flash-priority dataset export passes its training gate. They do not wait for the embedding caches to finish. The pre-submitted evaluation allocation depends on successful completion of both full fine-tunes and evaluates identical Flash test clips before/after tuning, fixed 2,000-clip P3 validation and test selections before/after tuning, the five public benchmarks, and matched actor-disjoint classification adapters.
', 'Embedding probes start independently as soon as their own Ladder/P3 training features are complete. Their evaluations wait for that model’s complete benchmark features. Gemini probe tuning starts when its legacy head and final Flash features are ready. Before full Whisper training finishes, two encoder-cache nodes run alongside two dedicated Whisper training nodes and one probe/evaluation node. Afterwards, the freed capacity permits up to four encoder-cache nodes; pending/running evaluations reduce that limit as needed. Maximum concurrency remains five study nodes, with four GPUs per node. Held jobs and unmet queue dependencies reserve no compute node.
', 'To fit earlier scheduler openings, full Whisper fine-tuning and evaluation jobs accept time windows as short as one hour; probe training accepts windows as short as 45 minutes, and encoder caching as short as two hours. These are allocation limits, not promised completion times. Training saves model, optimizer and scheduler states at epoch boundaries; probes additionally save their random states. Interrupted epochs restart from the last completed epoch; caches skip every committed chunk. Queue priority and the remaining provider batches still determine actual start times.
'] progress=[] for path in sorted((ROOT/'probes').rglob('metrics.jsonl')): completed=[] for line in path.read_text().splitlines(): try:completed.append(json.loads(line)) except json.JSONDecodeError:continue if completed: row=completed[-1];validation=row.get('validation',{}) progress.append([str(path.parent.relative_to(ROOT/'probes')),row['stage'],row['epoch'], row['update'],number(validation.get('loss')),validation.get('clips','Unknown')]) for size in ('base','small'): path=GEMINI/'training'/('whisper_'+size)/'metrics.jsonl' if path.exists(): for line in path.read_text().splitlines()[::-1]: try:row=json.loads(line) except json.JSONDecodeError:continue progress.append(['Gemini full FT / Whisper '+size,'Gemini',row['epoch'],row['update'], number(row.get('validation_loss')),'See fine-tuning config']);break if progress: parts += ['These are recorded epoch checks, before the final held-out benchmark evaluation. The loss combines masked normalized regression, speaker embedding and burst objectives with different weights; it is not accuracy, correlation or a paper benchmark score. Compare final per-target and public benchmark metrics below once they exist.
'] retention=[] for size in ('base','small'): out=ROOT/'whisper/gemini'/size for domain,before_file,after_file in [('Flash test','legacy_on_flash_test_metrics.json','gemini_test_metrics.json'), ('P3 validation','legacy_p3_validation_metrics.json','gemini_p3_validation_metrics.json'), ('P3 test','legacy_p3_test_metrics.json','gemini_p3_test_metrics.json')]: before=optional(out/before_file);after=optional(out/after_file) if before and after: retention.append([size,domain,after['clips'],number(before.get('frame_f1')),number(after.get('frame_f1')), number(after['frame_f1']-before['frame_f1'])]) if retention: parts += ['Frame F1 uses a fixed 0.5 decision threshold on the 20 ms grid. Flash test labels are machine-generated event annotations; P3 labels are known synthetic construction intervals. These domains have different labels and acoustic distributions. The current fine-tunes improve agreement with Flash targets while reducing performance on P3. Checkpoints were selected by Flash validation loss, and this two-epoch tuning stage contained no P3 replay. The separate localization IoU and boundary-error metrics below must also be considered; frame F1 alone does not measure event localization accuracy.
'] model_rows=[] for m in cfg['models']: source=m.get('repo',m.get('model_name')) jobs=workflow.get('cache_jobs',{}).get(m['id'],[]) temporal='160 ms patches' if m['backend']=='clap' else '20 ms' if m['backend']=='commercial' else 'Native audio-token grid; verified by cached token counts' model_rows.append([m['id'],source,m['native_dim'],temporal,', '.join(str(j['id']) for j in jobs) or 'Not submitted', workflow.get('cache_states',{}).get(m['id'],'Pending')]) parts += ['All nine embedding backbones stay frozen. Audio is resampled to mono 16 kHz and cropped to 30 seconds where necessary; NaFlex performs its native 32 kHz feature transform. No reference transcription, caption, ASR decoding or generated answer enters a probe. The exact epoch-320 CLAP checkpoints are used; Large is the 32-node checkpoint. Native embeddings and time-indexed features are saved once in immutable TAR sidecars and reused.
', 'Training-only PCA maps pooled embeddings to 256 dimensions and scales components using training statistics. XXS has 224 native dimensions: 224 retained components are padded with zeros to 256. Temporal features use a fixed seeded projection to 64 dimensions. Interpolation onto the common 20 ms target grid does not improve a backbone’s native timing resolution.
', 'The 192-score schema contains 40 EmoNet emotion targets, 57 VoiceNet dimensions, genuineness/blend/quality, Empathic Insight Plus outputs, AudioBox scores, all DNSMOS outputs, burst count and additional VoiceCLAP attributes. CPS is an independent 193rd scalar. Missing or out-of-domain teacher targets are masked; they are not replaced with zero labels. Gemini Flash values override older teacher values wherever a valid corresponding Flash annotation exists. Orange vectors come from their audio models; Gemini does not synthesize embeddings.
', 'The original Whisper encoders were trained over S1–S10 and have larger heads and a different training budget. Identical probe inputs/head sizes control downstream capacity across embedding backbones; they do not equalize backbone size, pretraining data or previous Whisper exposure. Ladder holdouts here are withheld from the new probes, but may have been seen by the previously trained Whisper model. Underlying Emolia source overlap and upstream benchmark exposure are not audited. Before/after Gemini scores use exactly the same Flash test clips; improvements combine extra training and changed targets.
', 'VoiceNet-Emo is emolia-emo; VoiceNet-Ext is emolia-dim, not an additional independent benchmark. We report the current repository ≥2-rater cut and unflagged cut, threshold-free mean per-prompt Spearman, the paper-style oracle threshold statistic and a separate five-fold clip-grouped threshold audit. Oracle thresholds fit the evaluated labels and must not be presented as held-out classification. The current Ext snapshot’s rater counts differ from the paper; see the existing detailed report.
EmoNet human intensities 0/1/2 become 0/5/10; native emotion outputs use a fixed 2.5 endpoint conversion without fitting benchmark labels. Forty mapped emotion categories cover 12,000 of the released 12,600 clips. CREMA-D and RAVDESS fixed taxonomy mappings are reported separately from supervised adapters. Each new adapter uses all 192 predictions, exactly the same parameter count across models, five outer actor-disjoint folds and three inner actor-disjoint folds to select LR (0.001 or 0.003) and epochs (20 or 50). Scaling is fitted inside each training fold. Every clip receives one outer held-out prediction; reported confidence intervals resample actors.
'] outputs=[] baseline=optional(BENCH/'public_scores.json').get('models',{}) transfer=[] for size in ('base','small'): previous=baseline.get(size,{}) tuned=optional(ROOT/'whisper/gemini'/size/'public_metrics.json') for benchmark,key,metric,label in [ ('emolia-emo','emolia_emo','mean_prompt_spearman','Mean per-prompt Spearman'), ('emolia-dim','emolia_dim','mean_prompt_spearman','Mean per-prompt Spearman')]: before=previous.get(key,{}).get('repo_min_2_raters',{}) after=tuned.get(benchmark,{}).get('repo_min_2_raters',{}) if before and after: transfer.append([size,benchmark,'Repository ≥2 raters; '+str(after['n'])+' pairs',label, number(before.get(metric)),number(after.get(metric))]) for benchmark in ('crema','ravdess'): before=optional(ROOT/'whisper/legacy'/size/(benchmark+'_matched_adapter.json')).get('metrics',{}) after=optional(ROOT/'whisper/gemini'/size/(benchmark+'_matched_adapter.json')).get('metrics',{}) if before and after: transfer.append([size,benchmark,'Same nested actor-disjoint folds','Matched score-MLP accuracy', number(before.get('accuracy')),number(after.get('accuracy'))]) if transfer: parts += ['The original rows reuse existing benchmark predictions for Base S4 and Small S3. The tuned rows evaluate their new full fine-tunes. The two Emolia correlations use the same repository ≥2-rater selection and require no fitted decision threshold. CREMA-D and RAVDESS use supervised score adapters with the same architecture and nested actor-disjoint protocol before and after tuning. Results are mixed: emotion correlations improve, CREMA-D is nearly unchanged for Base and slightly higher for Small, while RAVDESS adapter accuracy decreases for both. These measurements do not establish a uniform improvement across tasks.
'] reference=[] for name in ('base','small'): result=baseline.get(name,{}) emo=result.get('emolia_emo',{}).get('repo_min_2_raters',{}) dim=result.get('emolia_dim',{}).get('repo_min_2_raters',{}) en=result.get('emonet',{}).get('all_mapped_40',{}) reference.append(['Original Whisper '+name,number(emo.get('balanced_accuracy_oracle_per_prompt')), number(emo.get('mean_prompt_spearman')),number(dim.get('balanced_accuracy_oracle_per_prompt')), number(dim.get('mean_prompt_spearman')),number(en.get('pearson')),'Existing regression-head evaluation']) reference.extend([ ['VoiceCLAP-Small, VoiceNet paper','0.6754','0.3176','0.6367','0.1051','Different EmoNet protocol','Paper embedding/text-prompt evaluation'], ['VoiceCLAP-Large, VoiceNet paper','0.7021','0.3719','0.6510','0.1475','Different EmoNet protocol','Paper embedding/text-prompt evaluation'], ['VoiceCLAP-Large-v2, released model card','0.7069','0.3865','0.6816','0.2125','Not specified here','Released embedding benchmark numbers, not the new probes']]) parts += ['These Whisper rows predate Gemini tuning and this embedding-probe study. Paper and model-card rows are quoted as context from their linked primary sources; current benchmark snapshots, regression scoring versus text-prompt similarity, and upstream training exposure differ. These are not controlled same-protocol rankings. The newly trained probes will appear separately below.
'] for phase in ('legacy','gemini'): for m in cfg['models']: for variant in ('linear','mlp'): out=ROOT/'probes'/phase/m['id']/variant outputs.append((phase+' / '+m['id']+' / '+variant,out)) for phase in ('legacy','gemini'): for m in ('base','small'): outputs.append((phase+' / Whisper '+m,ROOT/'whisper'/phase/m)) rows=[] details=[] for name,out in outputs: public=optional(out/'public_metrics.json') internal=optional(out/'test_metrics.json') or optional(out/'gemini_test_metrics.json') domain='Flash test' if name.startswith('gemini') else 'Ladder S8–S10 + P3 test' if name.startswith('legacy / Whisper '): size=name.rsplit(' ',1)[-1] old=baseline.get(size,{}) public=public or {key:old[source] for key,source in [ ('emolia-emo','emolia_emo'),('emolia-dim','emolia_dim'), ('emonet','emonet'),('crema','crema'),('ravdess','ravdess')] if source in old} internal=internal or optional(ROOT/'whisper/gemini'/size/'legacy_on_flash_test_metrics.json') domain='Flash test, original checkpoint' emo=public.get('emolia-emo',{}).get('repo_min_2_raters',{}) dim=public.get('emolia-dim',{}).get('repo_min_2_raters',{}) emonet=public.get('emonet',{}).get('all_mapped_40',{}) adapters=[optional(out/(kind+'_matched_adapter.json')).get('metrics',{}) for kind in ('crema','ravdess')] rows.append([name,'Complete' if public else 'Pending',number(emo.get('balanced_accuracy_oracle_per_prompt')), number(emo.get('mean_prompt_spearman')),number(dim.get('mean_prompt_spearman')), number(adapters[0].get('accuracy')),number(adapters[1].get('accuracy')),domain,number(internal.get('frame_f1'))]) if internal or public: details.append(''+esc(json.dumps(public,indent=2))+'
'+esc(json.dumps({k:v for k,v in result.items() if k!='per_score'},indent=2))+'Internal frame F1 must be compared within the same test domain. Legacy embedding probes use the mixed Ladder/P3 holdout; Gemini runs use Flash test labels. Original and tuned Whisper rows both use the same Flash test clips; their separate P3 retention audit is above.
', 'Pending means the experiment has not produced a complete evaluation artifact. None of the pending rows are estimates. Supervised actor-CV adapter accuracy cannot be directly ranked against zero-shot paper accuracy.
',*details, 'VoiceNet paper · EmoNet-Voice paper · Human benchmark labels and taxonomy · Empathic Insight Voice Plus.
Study artifacts: '+esc(ROOT)+'. Machine-readable protocol: study.json. Source code: embedding-probe-code; Gemini full fine-tuning code and exact configurations. Report and new study code: CC BY 4.0, LAION; upstream models and datasets retain their own licenses.
'] style='body{font:16px/1.6 system-ui;background:#f3f6fa;color:#172b40;margin:0}main{max-width:1400px;margin:auto;padding:36px}h1{font-size:38px;line-height:1.2}h2{margin-top:36px}.lead{font-size:20px}.scroll{overflow:auto}table{border-collapse:collapse;width:100%;background:white;font-size:14px}th,td{padding:10px;border-bottom:1px solid #dde4ed;text-align:left;vertical-align:top}th{background:#e0ebf5}details{background:white;padding:14px;margin:12px 0}pre{white-space:pre-wrap;max-height:600px;overflow:auto}a{color:#075a9d}' OUT.write_text('