diff --git a/.gitattributes b/.gitattributes index f126de2a3ac3dd1266ab8f947c51988ee5671144..0fac566fb3b788a2d95442d0413c869ddb6f3ccf 100644 --- a/.gitattributes +++ b/.gitattributes @@ -128,3 +128,9 @@ samples/step_0190000/02-question.wav filter=lfs diff=lfs merge=lfs -text samples/step_0190000/03-numbers.wav filter=lfs diff=lfs merge=lfs -text samples/step_0190000/04-conversational.wav filter=lfs diff=lfs merge=lfs -text samples/step_0190000/05-long.wav filter=lfs diff=lfs merge=lfs -text +samples/prompt_choice/old_output/02-question.wav filter=lfs diff=lfs merge=lfs -text +samples/prompt_choice/old_output/03-numbers.wav filter=lfs diff=lfs merge=lfs -text +samples/prompt_choice/old_output/05-long.wav filter=lfs diff=lfs merge=lfs -text +samples/prompt_choice/old_prompt/02-question.wav filter=lfs diff=lfs merge=lfs -text +samples/prompt_choice/old_prompt/03-numbers.wav filter=lfs diff=lfs merge=lfs -text +samples/prompt_choice/old_prompt/05-long.wav filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md index fb0e448af671db324008199de787d82c8de7af45..46d8bc31915cadc4b98a7eedd4a3e04ba9fc5f0f 100644 --- a/README.md +++ b/README.md @@ -31,295 +31,93 @@ The name: **DAC** for the Semantic-DACVAE audio codec whose latents it generates English, **10k** for its training data, about ten thousand hours of speech. The code: [kadirnar/dacvae-next](https://github.com/kadirnar/dacvae-next/tree/roadmap/en-echo), Python package `mytts`. -> **Status: training, step 190k of 200k.** A new checkpoint (a copy of the model saved at that point of its training) -> appears here every 10k steps (from step 20k on), with its samples and scores: this page updates itself. Early checkpoints sound rougher than -> later ones. +> **Status: training is finished. The model is step 100k** (the EMA weights saved at that step). The run was planned for 200k steps +> and stopped at step 194,271 on 2026-10-04; this page no longer changes by itself. Why step 100k: later checkpoints sound a little +> better on held-out voices of the training data's kind, but copy real people's voices much worse. Speaker similarity on real voices +> (seed-dev SIM-o) falls from 0.50 at step 100k to 0.22 at step 190k, where 51 % of the outputs score below 0.2 (5.7 % at step +> 100k), with the same settings. Step 100k, with each prompt's own bandwidth as the condition (auto), is the best balance. Details: +> [#177](https://github.com/kadirnar/dacvae-next/issues/177). **On this page:** [Listen](#listen) · [Results](#results) · [How to use](#how-to-use) · [Training](#training) · [Data](#data) · [Licence](#licence) ## Listen -The newest checkpoint, **step 190k**, reads five texts. Each text is spoken in a different voice, copied from -the short voice prompt next to it. The model never heard these voices in training (held-out voices of the provided data). +The model, **step 100k**, reads five texts. Each text is spoken in a different voice, copied from the short voice prompt next +to it. The model never heard these voices in training (held-out voices of the provided data). - - - - - - + + + + + +
What the model readsVoice prompt (the input)Model output, step 190k
Short sentence
I left my umbrella at the office again, so I'm definitely getting soaked on the way home.

A higher voice (~202 Hz). It says: “I keep telling myself I just need to power through, like, push a little harder, and then it'll be fine. But it's not getting fine. It's getting— it's getting worse.”
Question
Have you ever noticed that the quietest person in the room usually has the most interesting story to tell?

A lower voice (~149 Hz). It says: “Wait, that mole— has it always looked like that? No, stop.”
Numbers, dates, abbreviations
Dr. Patel moved my appointment to Tuesday, March 3rd, at 4:15 p.m., and the co-pay went up from $20 to $35.

A higher voice (~181 Hz). It says: “this woman on the train was literally eating a whole bag of chips, like, crunching so loud, and I'm sitting there like, okay, do I move? whatever.”
Conversation (~10 s)
So I finally tried that new ramen place downtown, and honestly, it was worth the wait. The broth was amazing, but next time I'm definitely skipping the extra spicy option.

A lower voice (~115 Hz). It says: “I'm sorry, but the subway doors closed on my bag. Again.”
Long passage (~20 s)
When the storm passed, the whole neighborhood came outside to look at the damage, and although a few fences had fallen and the old oak tree had lost its biggest branch, everyone was relieved that nobody was hurt. They spent the rest of the afternoon clearing the street and sharing whatever food they had left.

The deepest voice (~108 Hz). It says: “Honestly I think we need to just call IT and have them... you know, actually fix it this time. I'm too old for this.”
What the model readsVoice prompt (the input)Model output (step 100k)
Short sentence
I left my umbrella at the office again, so I'm definitely getting soaked on the way home.

A higher voice (~202 Hz). It says: “I keep telling myself I just need to power through, like, push a little harder, and then it'll be fine. But it's not getting fine. It's getting— it's getting worse.”
Question
Have you ever noticed that the quietest person in the room usually has the most interesting story to tell?

A lower voice (~119 Hz). It says: “Look, I'm not saying it's easy, but you always pull it together. Just take it step by step, you know?”
Numbers, dates, abbreviations
Dr. Patel moved my appointment to Tuesday, March 3rd, at 4:15 p.m., and the co-pay went up from $20 to $35.

A higher voice (~201 Hz). It says: “Would you just— I know you mean well, but every time you mention it I feel like an idiot. It's probably nothing.”
Conversation (~10 s)
So I finally tried that new ramen place downtown, and honestly, it was worth the wait. The broth was amazing, but next time I'm definitely skipping the extra spicy option.

A lower voice (~115 Hz). It says: “I'm sorry, but the subway doors closed on my bag. Again.”
Long passage (~20 s)
When the storm passed, the whole neighborhood came outside to look at the damage, and although a few fences had fallen and the old oak tree had lost its biggest branch, everyone was relieved that nobody was hurt. They spent the rest of the afternoon clearing the street and sharing whatever food they had left.

The deepest voice (~102 Hz). It says: “I mean, come on, it's not like I woke up and decided to have the worst day ever. Things just went wrong one after another, like dominoes!”
-Older checkpoints (17): the same five texts and voices. Click one to open it. - -
-Step 180k: echo-dev WER 1.04 % · UTMOS 3.98 - - - - - - - - -
What the model readsModel output, step 180k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 170k: echo-dev WER 1.01 % · UTMOS 3.97 - - - - - - - - -
What the model readsModel output, step 170k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 160k: echo-dev WER 0.81 % · UTMOS 3.98 - - - - - - - - -
What the model readsModel output, step 160k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 150k: echo-dev WER 0.53 % · UTMOS 3.93 - - - - - - - - -
What the model readsModel output, step 150k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 140k: echo-dev WER 0.58 % · UTMOS 3.92 - - - - - - - - -
What the model readsModel output, step 140k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 130k: echo-dev WER 0.76 % · UTMOS 3.93 - - - - - - - - -
What the model readsModel output, step 130k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 120k: echo-dev WER 0.53 % · UTMOS 3.89 - - - - - - - - -
What the model readsModel output, step 120k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 110k: echo-dev WER 1.10 % · UTMOS 3.86 - - - - - - - - -
What the model readsModel output, step 110k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 100k: echo-dev WER 0.69 % · UTMOS 3.85 - - - - - - - - -
What the model readsModel output, step 100k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 90k: echo-dev WER 0.74 % · UTMOS 3.78 - - - - - - - - -
What the model readsModel output, step 90k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 80k: echo-dev WER 0.78 % · UTMOS 3.73 - - - - - - - - -
What the model readsModel output, step 80k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 70k: echo-dev WER 0.67 % · UTMOS 3.69 - - - - - - - - -
What the model readsModel output, step 70k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 60k: echo-dev WER 0.69 % · UTMOS 3.62 - - - - - - - - -
What the model readsModel output, step 60k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
- -
-Step 50k: echo-dev WER 0.74 % · UTMOS 3.56 - - - - - - - - -
What the model readsModel output, step 50k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
+How the samples are made: one take per text, no cherry-picking, with the settings of [How to use](#how-to-use) (32 Euler steps + sway, joint CFG w 4.0, initial noise 0.9, at most -16 LUFS (peak-limited), duration by the band rule, each prompt's own bandwidth as the condition (auto, clamped to 9,798-16,839 Hz)). +Every output carries an inaudible [AudioSeal](https://huggingface.co/facebook/audioseal) watermark, checked before upload; the prompts +are the original recordings. -
+### How the sample voices were chosen -
-Step 40k: echo-dev WER 0.74 % · UTMOS 3.50 +The voice prompt decides most of how an output sounds. In a test of this model on 45 held-out voices, 10 texts each, the choice of +prompt explained about 63 % of the differences in predicted naturalness (UTMOS), the text about 4 %. The model also copies the +recording: a prompt with a narrow frequency band (a dull, muffled recording) or with background noise gives an output that sounds +the same. So three of the five prompts above were replaced by clips that were checked first: held-out voices of the provided data +that pass the same prompt rules, recorded with a wide band and no background noise (measured on the prompts). Below, the model +(step 100k) reads the same text with the old and the new prompt (same settings and seed), with each output's band and predicted +naturalness. Details: [#177](https://github.com/kadirnar/dacvae-next/issues/177). - - - - - - + + + +
What the model readsModel output, step 40k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
What the model readsOld voice promptOutput with itNew voice promptOutput with it
Question
Replaced: a dull recording (band 8.9 kHz)

spk_0342 · band 8.9 kHz

band 8.2 kHz · UTMOS 3.84

spk_0327 · band 17.0 kHz

band 15.0 kHz · UTMOS 4.25
Numbers, dates, abbreviations
Replaced: background noise (signal-to-noise 38 dB)

spk_2669 · band 17.7 kHz

band 17.5 kHz · UTMOS 3.70

spk_0409 · band 15.5 kHz

band 15.3 kHz · UTMOS 4.38
Long passage (~20 s)
Replaced: a dull recording (band 10.9 kHz)

spk_1437 · band 10.9 kHz

band 8.3 kHz · UTMOS 4.15

spk_1964 · band 15.7 kHz

band 14.5 kHz · UTMOS 4.38
-
- -
-Step 30k: echo-dev WER 1.06 % · UTMOS 3.43 +Band: the highest frequency with real content in the recording (the estimator of the training data). UTMOS: predicted +naturalness, 1 to 5. Both measured on the take of the same text in the sweep of [#177](https://github.com/kadirnar/dacvae-next/issues/177) (same model, settings +and seed); the players hold this page's own takes, which can differ slightly. - - - - - - - -
What the model readsModel output, step 30k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
- -
+## Results -
-Step 20k: echo-dev WER 1.20 % · UTMOS 3.33 +### The model: step 100k - - - - - - - -
What the model readsModel output, step 20k
Short sentence
Question
Numbers, dates, abbreviations
Conversation (~10 s)
Long passage (~20 s)
+How well the model does on voices and sentences it never saw in training (**lower WER** is better, **higher SIM-o and UTMOS** are +better). The settings are those of [How to use](#how-to-use): each prompt's own bandwidth as the condition (auto, clamped to 9,798-16,839 Hz). -
+| Test set | WER ↓ | SIM-o ↑ | UTMOS ↑ | +|---|---:|---:|---:| +| **seed-dev**: real people's voices (545 sentences, two takes each) | 1.42 % | 0.526 | 3.80 (real speech: 3.52) | +| **echo-dev v2**: held-out voices of the training data's kind (198 voices, 1,000 sentences, one take each) | 0.53 % | 0.816 | 3.99 | -How the samples are made: one take per text, no cherry-picking, with the settings of [How to use](#how-to-use) (32 Euler steps + sway, joint CFG w 4.0, initial noise 0.9, at most -16 LUFS (peak-limited), duration by the band rule). -Every output carries an inaudible [AudioSeal](https://huggingface.co/facebook/audioseal) watermark, checked before upload; the prompts -are the original recordings. +2.4 % of the seed-dev outputs score SIM-o below 0.2 (the voice is not copied). +For comparison, step 190k with the default bandwidth condition (auto was not measured there): seed-dev WER 2.06 % · SIM-o 0.223 · UTMOS 3.00, 51 % below 0.2; echo-dev v2 WER 0.46 % · SIM-o 0.820 · UTMOS 4.14. -## Results +### All checkpoints of the run -How well each checkpoint does on voices and sentences it never saw in training (**lower WER** is better, **higher SIM-o and -UTMOS** are better): +Every checkpoint of the training run, scored with the default bandwidth condition; the model is step 100k: | Checkpoint | echo-dev WER ↓ | echo-dev SIM-o ↑ | echo-dev UTMOS ↑ (real speech: 4.20) | seed-dev WER ↓ | seed-dev SIM-o ↑ | seed-dev UTMOS ↑ (real speech: 3.52) | Samples | Weights | |---|---:|---:|---:|---:|---:|---:|---|---| -| **step 190k** (newest) | 0.64 % | 0.790 | 3.98 | - | - | - | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0190000) | -| step 180k | 1.04 % | 0.792 | 3.98 | 1.65 % | 0.261 | 3.29 | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0180000) | -| step 170k | 1.01 % | 0.796 | 3.97 | - | - | - | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0170000) | -| step 160k | 0.81 % | 0.796 | 3.98 | 1.58 % | 0.349 | 3.66 | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0160000) | -| step 150k | 0.53 % | 0.794 | 3.93 | - | - | - | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0150000) | -| step 140k | 0.58 % | 0.793 | 3.92 | 1.60 % | 0.404 | 3.81 | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0140000) | -| step 130k | 0.76 % | 0.792 | 3.93 | - | - | - | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0130000) | -| step 120k | 0.53 % | 0.790 | 3.89 | 1.45 % | 0.451 | 3.86 | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0120000) | -| step 110k | 1.10 % | 0.789 | 3.86 | - | - | - | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0110000) | -| step 100k | 0.69 % | 0.787 | 3.85 | 1.54 % | 0.499 | 3.78 | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0100000) | -| step 90k | 0.74 % | 0.782 | 3.78 | - | - | - | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0090000) | -| step 80k | 0.78 % | 0.776 | 3.73 | 1.58 % | 0.507 | 3.68 | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0080000) | -| step 70k | 0.67 % | 0.766 | 3.69 | - | - | - | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0070000) | -| step 60k | 0.69 % | 0.759 | 3.62 | 1.61 % | 0.537 | 3.56 | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0060000) | -| step 50k | 0.74 % | 0.754 | 3.56 | - | - | - | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0050000) | -| step 40k | 0.74 % | 0.748 | 3.50 | 1.85 % | 0.536 | 3.48 | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0040000) | -| step 30k | 1.06 % | 0.741 | 3.43 | - | - | - | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0030000) | -| step 20k | 1.20 % | 0.722 | 3.33 | 2.47 % | 0.506 | 3.27 | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0020000) | +| step 190k | 0.64 % | 0.790 | 3.98 | - | - | - | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0190000) | +| step 180k | 1.04 % | 0.792 | 3.98 | 1.65 % | 0.261 | 3.29 | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0180000) | +| step 170k | 1.01 % | 0.796 | 3.97 | - | - | - | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0170000) | +| step 160k | 0.81 % | 0.796 | 3.98 | 1.58 % | 0.349 | 3.66 | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0160000) | +| step 150k | 0.53 % | 0.794 | 3.93 | - | - | - | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0150000) | +| step 140k | 0.58 % | 0.793 | 3.92 | 1.60 % | 0.404 | 3.81 | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0140000) | +| step 130k | 0.76 % | 0.792 | 3.93 | - | - | - | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0130000) | +| step 120k | 0.53 % | 0.790 | 3.89 | 1.45 % | 0.451 | 3.86 | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0120000) | +| step 110k | 1.10 % | 0.789 | 3.86 | - | - | - | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0110000) | +| **step 100k** (the model) | 0.69 % | 0.787 | 3.85 | 1.54 % | 0.499 | 3.78 | [listen](#listen) | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0100000) | +| step 90k | 0.74 % | 0.782 | 3.78 | - | - | - | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0090000) | +| step 80k | 0.78 % | 0.776 | 3.73 | 1.58 % | 0.507 | 3.68 | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0080000) | +| step 70k | 0.67 % | 0.766 | 3.69 | - | - | - | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0070000) | +| step 60k | 0.69 % | 0.759 | 3.62 | 1.61 % | 0.537 | 3.56 | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0060000) | +| step 50k | 0.74 % | 0.754 | 3.56 | - | - | - | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0050000) | +| step 40k | 0.74 % | 0.748 | 3.50 | 1.85 % | 0.536 | 3.48 | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0040000) | +| step 30k | 1.06 % | 0.741 | 3.43 | - | - | - | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0030000) | +| step 20k | 1.20 % | 0.722 | 3.33 | 2.47 % | 0.506 | 3.27 | pending | [files](https://huggingface.co/VoiceHub/DACFlow-EN-10k/tree/main/checkpoints/step_0020000) | - **WER** (word error rate): Whisper-large-v3 writes down what it hears and this is compared with the text; 2 % means about one word in fifty is wrong. @@ -329,8 +127,10 @@ UTMOS** are better): - **echo-dev**: held-out voices of the training data's kind (synthetic EchoTTS voices, none of them trained on). **seed-dev**: real people's voices (the dev half of Seed-TTS test-en, Common Voice recordings). seed-dev is harder: the model learned only from synthetic voices. +- **echo-dev v2**: a larger set of the same kind (198 held-out voices), the echo set the model was chosen on; the table of all + checkpoints uses the first echo-dev set. - `-`: not measured at that step (echo-dev is scored every 10k steps, seed-dev every 20k). Every number averages two takes - per sentence. + per sentence (echo-dev v2: one take per sentence). ## How to use @@ -339,6 +139,7 @@ UTMOS** are better): ```bash git clone -b roadmap/en-echo https://github.com/kadirnar/dacvae-next cd dacvae-next +git checkout 029dec6 # the code that made the samples above pip install -e ".[codec]" ``` @@ -352,9 +153,9 @@ from huggingface_hub import snapshot_download from mytts.flow.sampler import SamplerConfig from mytts.infer import Synthesizer -ckpt = "checkpoints/step_0190000" # the newest checkpoint; any row of the Results table works +ckpt = "checkpoints/step_0100000" # the model (step 100k) local = snapshot_download("VoiceHub/DACFlow-EN-10k", allow_patterns=[f"{ckpt}/*"]) # downloads only this checkpoint -tts = Synthesizer.from_export(f"{local}/{ckpt}", device="cuda") +tts = Synthesizer.from_export(f"{local}/{ckpt}", device="cuda", bandwidth_range=(9797.6, 16839.0)) sc = SamplerConfig(steps=32, cfg_mode="joint", cfg_w=4.0, noise_scale=0.9) # the settings of the samples above wav, sr = tts.synthesize( "Any English text you like.", @@ -362,21 +163,27 @@ wav, sr = tts.synthesize( sc=sc, prompt_audio="my_voice.wav", # the voice to copy prompt_text="The exact words spoken in my_voice.wav.", + bandwidth_hz="auto", # condition on the prompt's own bandwidth (clamped to the range above) out_lufs=-16.0, seed=0, ) sf.write("output.wav", wav, sr) # 48 kHz, with the AudioSeal watermark ``` -**Or from the command line** (these are its default settings): +**Or from the command line**: ```bash -hf download VoiceHub/DACFlow-EN-10k --include "checkpoints/step_0190000/*" --local-dir DACFlow-EN-10k -python scripts/synthesize.py --model DACFlow-EN-10k/checkpoints/step_0190000 \ +hf download VoiceHub/DACFlow-EN-10k --include "checkpoints/step_0100000/*" --local-dir DACFlow-EN-10k +python scripts/synthesize.py --model DACFlow-EN-10k/checkpoints/step_0100000 --bandwidth auto --bandwidth-range 9797.6 16839.0 \ --text "Any English text you like." --prompt-audio my_voice.wav \ --prompt-text "The exact words spoken in my_voice.wav." --out output.wav ``` +`auto` measures the voice prompt's frequency band and asks the model for an output of the same band (a dull prompt +gives a dull output; see [How the sample voices were chosen](#how-the-sample-voices-were-chosen)). +The range is the 5th to 95th percentile of the training data's band; it is passed explicitly because a released +`config.json` records only the top of it. `auto` needs the code of [PR #180](https://github.com/kadirnar/dacvae-next/pull/180) or later (the commit above has it). + Tips: a clean prompt with an exact transcript works best. A long text is split at sentence ends and read piece by piece. `python scripts/watermark_check.py detect output.wav` checks the watermark. @@ -386,7 +193,8 @@ Tips: a clean prompt with an exact transcript works best. A long text is split a generates the latents of the [Semantic-DACVAE](https://huggingface.co/Aratako/Semantic-DACVAE-Japanese) audio codec (48 kHz, 25 frames per second) from the text, continuing the voice prompt in context (no separate speaker encoder). - **Data**: the curated training catalog `echo_en_v1` of the provided data ([EN-10, #34](https://github.com/kadirnar/dacvae-next/issues/34) curation; see [Data](#data)). -- **Recipe**: 200k steps on one RTX 5090, about 27 minutes of speech per step (40,000 latent frames); learning rate 2.5e-4 after 5k warm-up steps, held, then lowered over the last 20 % of the steps (from step 160k). +- **Recipe**: planned 200k steps on one RTX 5090, about 27 minutes of speech per step (40,000 latent frames); learning rate 2.5e-4 after 5k warm-up steps, held, then lowered over the last 20 % of the steps (from step 160k). + Training was stopped at step 194,271 on 2026-10-04. The model is the EMA export of step 100k, before the learning rate was lowered. The settings were chosen with small screening runs: the flow-matching noise schedule t_mean -0.8 / t_std 0.8 ([EN-55, #100](https://github.com/kadirnar/dacvae-next/issues/100)) and cross-utterance voice prompts with p_cross 0.6 ([EN-29, #53](https://github.com/kadirnar/dacvae-next/issues/53)). - **Code and history**: the recipe [configs/train/en_full.yaml](https://github.com/kadirnar/dacvae-next/blob/roadmap/en-echo/configs/train/en_full.yaml) diff --git a/prompts/02-question.wav b/prompts/02-question.wav index b102f4a2cabf4b888224e798091ad076e79e88eb..dfaabf010233fd5ebd05ea6b18fb41b7f432b677 100644 --- a/prompts/02-question.wav +++ b/prompts/02-question.wav @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:337e2af17f4096f6afa7beb9a151859157c3ee8ec3cbfeca9e63aa2a61529cab -size 380972 +oid sha256:ec5010cbb72864e62cd95a87840c6f74a750f4ac135ddcda43ce34417b63c346 +size 503852 diff --git a/prompts/03-numbers.wav b/prompts/03-numbers.wav index b1a56fa0dbd8c20d82c2ad3fa8b77080c9f9fbfc..21d15fbfa1760c23143be5482822a6e51111ee17 100644 --- a/prompts/03-numbers.wav +++ b/prompts/03-numbers.wav @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:c7949d82713134b3b4480cfe85cd599e5ddfc73e860a96de1b09bee04b9102fd +oid sha256:4f1d94263fbe54bdcd239706933b97365dac4d34ea62f75e319da06821cdeff7 size 634924 diff --git a/prompts/05-long.wav b/prompts/05-long.wav index 88a4d09dff3c16bb1397d15aece0c7163862fed0..6a95c3e4a7e1b730b1cdd2b008cd50d5532457ad 100644 --- a/prompts/05-long.wav +++ b/prompts/05-long.wav @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:b0047e2d525a230a7d05e70ac838100b92c425687e7cc0a3c5ab0e065c04af95 -size 622636 +oid sha256:9f6966023f8946ea8d9d915dbd3a8ac00c242217b131dcc979eed1063df2b9e0 +size 708652 diff --git a/samples/prompt_choice/old_output/02-question.wav b/samples/prompt_choice/old_output/02-question.wav new file mode 100644 index 0000000000000000000000000000000000000000..31e8efbdbc1d1f4bd28ef4d658014bc568c35919 --- /dev/null +++ b/samples/prompt_choice/old_output/02-question.wav @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0d251f50641106b27a64844cdaae6ddf68539441021cea535a1419ff1595be34 +size 675884 diff --git a/samples/prompt_choice/old_output/03-numbers.wav b/samples/prompt_choice/old_output/03-numbers.wav new file mode 100644 index 0000000000000000000000000000000000000000..2a90d0f0e35aed26b5e9dd85ad3b8c3d43648ee6 --- /dev/null +++ b/samples/prompt_choice/old_output/03-numbers.wav @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:45490e02ea584a23db51da972998d91353bd6204fc8ebab5adb609e28f6befc1 +size 802604 diff --git a/samples/prompt_choice/old_output/05-long.wav b/samples/prompt_choice/old_output/05-long.wav new file mode 100644 index 0000000000000000000000000000000000000000..403fd781ff84b656364a87b6d0014e2ca350b40f --- /dev/null +++ b/samples/prompt_choice/old_output/05-long.wav @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:20652f389b6b9a5b7424ee8dd42765e4aa33eaa6d0176812698226990fdba49e +size 1699244 diff --git a/samples/prompt_choice/old_prompt/02-question.wav b/samples/prompt_choice/old_prompt/02-question.wav new file mode 100644 index 0000000000000000000000000000000000000000..b102f4a2cabf4b888224e798091ad076e79e88eb --- /dev/null +++ b/samples/prompt_choice/old_prompt/02-question.wav @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:337e2af17f4096f6afa7beb9a151859157c3ee8ec3cbfeca9e63aa2a61529cab +size 380972 diff --git a/samples/prompt_choice/old_prompt/03-numbers.wav b/samples/prompt_choice/old_prompt/03-numbers.wav new file mode 100644 index 0000000000000000000000000000000000000000..b1a56fa0dbd8c20d82c2ad3fa8b77080c9f9fbfc --- /dev/null +++ b/samples/prompt_choice/old_prompt/03-numbers.wav @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c7949d82713134b3b4480cfe85cd599e5ddfc73e860a96de1b09bee04b9102fd +size 634924 diff --git a/samples/prompt_choice/old_prompt/05-long.wav b/samples/prompt_choice/old_prompt/05-long.wav new file mode 100644 index 0000000000000000000000000000000000000000..88a4d09dff3c16bb1397d15aece0c7163862fed0 --- /dev/null +++ b/samples/prompt_choice/old_prompt/05-long.wav @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b0047e2d525a230a7d05e70ac838100b92c425687e7cc0a3c5ab0e065c04af95 +size 622636 diff --git a/samples/step_0020000/01-short.wav b/samples/step_0020000/01-short.wav deleted file mode 100644 index 45a36615a9324bb653c218b4d01e01ed3720041e..0000000000000000000000000000000000000000 --- a/samples/step_0020000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:50735a6bf3454a57b37fb3ec0e543af1f0c3e2eb46b2fbd13e17619081ecf6f1 -size 491564 diff --git a/samples/step_0020000/02-question.wav b/samples/step_0020000/02-question.wav deleted file mode 100644 index a60a0b7089b514291a6716bf38aaf6394a600f18..0000000000000000000000000000000000000000 --- a/samples/step_0020000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:297d3d2284db20606a1c592847ec05c05c0cb6c3ac5eb7553762bd5ea64bcaec -size 675884 diff --git a/samples/step_0020000/03-numbers.wav b/samples/step_0020000/03-numbers.wav deleted file mode 100644 index 3a09bbfeb328812b86fe985682a47e637822594e..0000000000000000000000000000000000000000 --- a/samples/step_0020000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c125ca86b2c4234331edff603762bff5c2560c0279daa8ed0b3864acb1ba234e -size 802604 diff --git a/samples/step_0020000/04-conversational.wav b/samples/step_0020000/04-conversational.wav deleted file mode 100644 index e45e9155b3594d0d37593f70beefd664d842ff3f..0000000000000000000000000000000000000000 --- a/samples/step_0020000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e4bd481b97e04d9663bac0b8378c1aabf60a4876f2aebe2d23e30327049af680 -size 1129004 diff --git a/samples/step_0020000/05-long.wav b/samples/step_0020000/05-long.wav deleted file mode 100644 index 2511748dc3d47e5743c367a17b4e58e5f1fc4dfd..0000000000000000000000000000000000000000 --- a/samples/step_0020000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4841a5825a32e346403b0300b10ca8c52810384be26d19fa76e8bf79778a18f2 -size 1699244 diff --git a/samples/step_0030000/01-short.wav b/samples/step_0030000/01-short.wav deleted file mode 100644 index acd406e37c32461ecfdfcb590411e59973e3b04d..0000000000000000000000000000000000000000 --- a/samples/step_0030000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ef8a4af218c3446f88cc66c5a1685af66d3287a1dd4023d5a286e79cf8f5bfa0 -size 491564 diff --git a/samples/step_0030000/02-question.wav b/samples/step_0030000/02-question.wav deleted file mode 100644 index 1e5731645f58a90d2876ece8dc400368fc3c52d2..0000000000000000000000000000000000000000 --- a/samples/step_0030000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:30b85591e417f62ebe5ee75787e647ee7af5926de93184ad6e5679b9d29efe81 -size 675884 diff --git a/samples/step_0030000/03-numbers.wav b/samples/step_0030000/03-numbers.wav deleted file mode 100644 index 8196e8e686a3bb7ba120e75b9283ea2ea470fb63..0000000000000000000000000000000000000000 --- a/samples/step_0030000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:78d2072e93a23915d2693d7cc7aa86885e3aaf784136826fe4a09267607cf075 -size 802604 diff --git a/samples/step_0030000/04-conversational.wav b/samples/step_0030000/04-conversational.wav deleted file mode 100644 index 27cf49ac4e6aad573083c2a027b3ae5dccff08c4..0000000000000000000000000000000000000000 --- a/samples/step_0030000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:268121f68f855288f80d35f0dac735393ac2b7aeab9fe3c3239b1fd6e7031711 -size 1129004 diff --git a/samples/step_0030000/05-long.wav b/samples/step_0030000/05-long.wav deleted file mode 100644 index 344b4ce30f161ede09a1f99381041fd3207c9969..0000000000000000000000000000000000000000 --- a/samples/step_0030000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7d7d7dec305fcd3195d9f0d77c5aea0c033080778de8ef4edbe7e63880f26e84 -size 1699244 diff --git a/samples/step_0040000/01-short.wav b/samples/step_0040000/01-short.wav deleted file mode 100644 index 3283205d553d5eb6adf955966b6949aa0201057d..0000000000000000000000000000000000000000 --- a/samples/step_0040000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:49a9ad994685c295924654ab4bb913396c512f83fe8c61d611af31d41a109e2a -size 491564 diff --git a/samples/step_0040000/02-question.wav b/samples/step_0040000/02-question.wav deleted file mode 100644 index 842b1ca57d03c58e7f03426bb75976501bfc46b5..0000000000000000000000000000000000000000 --- a/samples/step_0040000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:191d8b96c3c567b1fe15eaa95505db8e15d06cf3f554063fbeecd855b006e53e -size 675884 diff --git a/samples/step_0040000/03-numbers.wav b/samples/step_0040000/03-numbers.wav deleted file mode 100644 index 1676a4a30ccc8dffec882bacee09a7db8ccb839b..0000000000000000000000000000000000000000 --- a/samples/step_0040000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8fb61e22ef35366165589193edf7e23b620652cd19b2ae646695c5f964f84267 -size 802604 diff --git a/samples/step_0040000/04-conversational.wav b/samples/step_0040000/04-conversational.wav deleted file mode 100644 index da7b9c064905084c2316ccf5a64bd202cde759b3..0000000000000000000000000000000000000000 --- a/samples/step_0040000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7ec715e56ca0035b5a992d05e424eee61738b3f4565d84e05cde53e3b94f1070 -size 1129004 diff --git a/samples/step_0040000/05-long.wav b/samples/step_0040000/05-long.wav deleted file mode 100644 index 1796e2ca000847978726c18ed93aa3de62e0c610..0000000000000000000000000000000000000000 --- a/samples/step_0040000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5525bb18667843cb8fec70df78f4f5478b79faef1a753e7ec7a02a11122a6691 -size 1699244 diff --git a/samples/step_0050000/01-short.wav b/samples/step_0050000/01-short.wav deleted file mode 100644 index 1cc9011b2c8838135cf38d5068a585b3d46d66c4..0000000000000000000000000000000000000000 --- a/samples/step_0050000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:90977dc5598fed3c02d897870e0661c045b25375ea5594e11991564d0c6112e2 -size 491564 diff --git a/samples/step_0050000/02-question.wav b/samples/step_0050000/02-question.wav deleted file mode 100644 index bf0661ea578e32d89be7fb9b8df8b517244fd2be..0000000000000000000000000000000000000000 --- a/samples/step_0050000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c11a29b4306ff85b6a9f1834389201a3a8eaa6dc4084dcac4a2130bb28c89eb9 -size 675884 diff --git a/samples/step_0050000/03-numbers.wav b/samples/step_0050000/03-numbers.wav deleted file mode 100644 index 459b5e4cbd6fd40189ed910755a9553c130bf964..0000000000000000000000000000000000000000 --- a/samples/step_0050000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:cd6993d65d2ef0ee2e695bed119d56fb0a3ec261dde9b126244814d0a0cb49c1 -size 802604 diff --git a/samples/step_0050000/04-conversational.wav b/samples/step_0050000/04-conversational.wav deleted file mode 100644 index 6101c91a928658be2e119f2ba2749627f27b1ea3..0000000000000000000000000000000000000000 --- a/samples/step_0050000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9f8c8b9578823d51830602bb7f3dafa2ac0639a4c647d43871d5ef7b6ed8c433 -size 1129004 diff --git a/samples/step_0050000/05-long.wav b/samples/step_0050000/05-long.wav deleted file mode 100644 index 596b7e71f5737ad4fc099f40629a1eb3d1115b51..0000000000000000000000000000000000000000 --- a/samples/step_0050000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3253fea011132cb064196f174d8856a924ecae7f3dedce0c75ab91192f0f7de5 -size 1699244 diff --git a/samples/step_0060000/01-short.wav b/samples/step_0060000/01-short.wav deleted file mode 100644 index 1adfc368da91ba010c047a2f78664a6da7c8d07f..0000000000000000000000000000000000000000 --- a/samples/step_0060000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:aa41e9039c23632e6e23dc63defe11549bf664c606ad8dd02a17de95f44a5954 -size 491564 diff --git a/samples/step_0060000/02-question.wav b/samples/step_0060000/02-question.wav deleted file mode 100644 index 55b77d981670f2fd864879d0466edde62874fc66..0000000000000000000000000000000000000000 --- a/samples/step_0060000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0302c8685d9e47b71c2fc891e7ae18475ce287747186d1f061a82c6b626d4ad6 -size 675884 diff --git a/samples/step_0060000/03-numbers.wav b/samples/step_0060000/03-numbers.wav deleted file mode 100644 index 7d982c5e49e3e8d64677572b5d7b9965488865c7..0000000000000000000000000000000000000000 --- a/samples/step_0060000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:21cd0b3472609ece712c22c0c12d72134387b17ac90ef5d9084f4b8725fd25e8 -size 802604 diff --git a/samples/step_0060000/04-conversational.wav b/samples/step_0060000/04-conversational.wav deleted file mode 100644 index 4fdf6b35a1e02d82091829cc3b93fce890d639dd..0000000000000000000000000000000000000000 --- a/samples/step_0060000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1c935be979bca4df237b0dc95c0f2188150c5e3b04324c4e702a02dbb1977d6b -size 1129004 diff --git a/samples/step_0060000/05-long.wav b/samples/step_0060000/05-long.wav deleted file mode 100644 index fc00cde6bdeecdf59f4a90d8f3328f153d695a49..0000000000000000000000000000000000000000 --- a/samples/step_0060000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4e2199443c386340cefcb37edc19c71a9ee2e5c10ffc97b4d5b6d0d994511165 -size 1699244 diff --git a/samples/step_0070000/01-short.wav b/samples/step_0070000/01-short.wav deleted file mode 100644 index 40e182ff94d80b46bf44a5507baf9559f39aa3af..0000000000000000000000000000000000000000 --- a/samples/step_0070000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:abec54dbe34fda663904f8c0266a7cb02519b557a6b64cd106ee7ad0f95916b1 -size 491564 diff --git a/samples/step_0070000/02-question.wav b/samples/step_0070000/02-question.wav deleted file mode 100644 index 4d26ec9b36b893da480f9ff5588b1b540977951c..0000000000000000000000000000000000000000 --- a/samples/step_0070000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:55da8e9b91b52837b0e508a0aef7e7de32ed47add1cf5faf6e6c63a1434932da -size 675884 diff --git a/samples/step_0070000/03-numbers.wav b/samples/step_0070000/03-numbers.wav deleted file mode 100644 index bda729c0514186e88e62222fedc11a719c86eeab..0000000000000000000000000000000000000000 --- a/samples/step_0070000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4c68b434f4d130f86610bfb7ed090a91595a7fe16072f7c6703ba7c032efd346 -size 802604 diff --git a/samples/step_0070000/04-conversational.wav b/samples/step_0070000/04-conversational.wav deleted file mode 100644 index b81e9a222aa7df6d7dd9cf3c6ff0b41d4ef37637..0000000000000000000000000000000000000000 --- a/samples/step_0070000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6437479879b21c243c6360a6270f7eb662d593315c8ebe79e07ad31190889435 -size 1129004 diff --git a/samples/step_0070000/05-long.wav b/samples/step_0070000/05-long.wav deleted file mode 100644 index b00933fbbc832f8f106016366aee99aecb279bb2..0000000000000000000000000000000000000000 --- a/samples/step_0070000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:be36c40ab725b3638cfdd159f68c283e0a38e19636335255ee5e3a8c1e819693 -size 1699244 diff --git a/samples/step_0080000/01-short.wav b/samples/step_0080000/01-short.wav deleted file mode 100644 index 66eeef3ee949c72e15c6f911d015fa8e6bebefbf..0000000000000000000000000000000000000000 --- a/samples/step_0080000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:220a06fd91bc5b539bfab4cc4cf6d7fa8bb207b1bd4a8136b0e9fa0b820ef99e -size 491564 diff --git a/samples/step_0080000/02-question.wav b/samples/step_0080000/02-question.wav deleted file mode 100644 index 0aafbaae52e08a825041669cc561f31e23fc94f7..0000000000000000000000000000000000000000 --- a/samples/step_0080000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fdf40c4150f1063d6b141721f574208fcb3b4b4a218044880e31cd1b9801b666 -size 675884 diff --git a/samples/step_0080000/03-numbers.wav b/samples/step_0080000/03-numbers.wav deleted file mode 100644 index 9b51c1539b53a691a6cd0431a2750b3739f0dfa0..0000000000000000000000000000000000000000 --- a/samples/step_0080000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f508b564c626e54bf17a85b9d1cde757a7865b63bb2c6438006cbf314422d9ce -size 802604 diff --git a/samples/step_0080000/04-conversational.wav b/samples/step_0080000/04-conversational.wav deleted file mode 100644 index ea1e412224497155eae924b26eca5827d0754c70..0000000000000000000000000000000000000000 --- a/samples/step_0080000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:297a1d51596503186a06986a690e9b899e360df647bcfd970b839a4636fcfcf4 -size 1129004 diff --git a/samples/step_0080000/05-long.wav b/samples/step_0080000/05-long.wav deleted file mode 100644 index a7e5a6e95e8ec29b5da8002e968ad691d76f1df7..0000000000000000000000000000000000000000 --- a/samples/step_0080000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7204173c814b5924fdd90afa379c8a42e31b3c54b4068f9a0063ea3393ea7539 -size 1699244 diff --git a/samples/step_0090000/01-short.wav b/samples/step_0090000/01-short.wav deleted file mode 100644 index 2541045f2d5262419f3cbb3e8ae23671b8ec45a0..0000000000000000000000000000000000000000 --- a/samples/step_0090000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:912169451f2095e83992c48b83187a6e6eaf20224fe36d55866ae4b339eaec16 -size 491564 diff --git a/samples/step_0090000/02-question.wav b/samples/step_0090000/02-question.wav deleted file mode 100644 index 5a24bfb647903d01b948391ba063a8179fa9c954..0000000000000000000000000000000000000000 --- a/samples/step_0090000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:331f3aa5925718dfd401626289b343560d11f02f84fb231c9cc884b783308162 -size 675884 diff --git a/samples/step_0090000/03-numbers.wav b/samples/step_0090000/03-numbers.wav deleted file mode 100644 index d6bb2160c1ba5922cb4733b57e54be353bc37c91..0000000000000000000000000000000000000000 --- a/samples/step_0090000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:eec195dfc8db1a8fa38e2c5e2e46f971aafdfe66eed096dce2f12d1bd7e7f87e -size 802604 diff --git a/samples/step_0090000/04-conversational.wav b/samples/step_0090000/04-conversational.wav deleted file mode 100644 index b667c1bf2e9b2cb743169cb4b0e14e01dafae064..0000000000000000000000000000000000000000 --- a/samples/step_0090000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9fad8387f13bb3e23258cb435f0aa0f08d8303929b683a559c83b88915fef1cb -size 1129004 diff --git a/samples/step_0090000/05-long.wav b/samples/step_0090000/05-long.wav deleted file mode 100644 index 8f99db339e8764052ab61adf740c6fc481b585fd..0000000000000000000000000000000000000000 --- a/samples/step_0090000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b372727c0000be7091c5c9d29ed10ca65c116edf98b81fc4099df47ca103d9ff -size 1699244 diff --git a/samples/step_0100000/01-short.wav b/samples/step_0100000/01-short.wav index d9e34c61f43bcc78a0464e228aec78663727ebf1..9ed2d56825a49f558f307f45f20a02a337ef2062 100644 --- a/samples/step_0100000/01-short.wav +++ b/samples/step_0100000/01-short.wav @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:07e3134dbff8e6f457985e581a86d768b8a87f109c0428148f97a8d7a0094f21 +oid sha256:be50f96dc34d32d24739068756b6f52e88e3b0234396ee501502ba6732c548f6 size 491564 diff --git a/samples/step_0100000/02-question.wav b/samples/step_0100000/02-question.wav index 896afeae592c40fb41934847fec064912427d6d6..47c315056fe8f2560a963b50e97934c30342484c 100644 --- a/samples/step_0100000/02-question.wav +++ b/samples/step_0100000/02-question.wav @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:2e391b9baba3493423e56561992e5a267a62b86d8844a7272d6e1ce43e661ab4 -size 675884 +oid sha256:1a64e3bbded645444bab1abce831686febe10ad2efc5a45d72227d2d2821fcb8 +size 583724 diff --git a/samples/step_0100000/03-numbers.wav b/samples/step_0100000/03-numbers.wav index 2a90d0f0e35aed26b5e9dd85ad3b8c3d43648ee6..fd79b93119df4385945b4d4e0a3f3ad16949b0ac 100644 --- a/samples/step_0100000/03-numbers.wav +++ b/samples/step_0100000/03-numbers.wav @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:45490e02ea584a23db51da972998d91353bd6204fc8ebab5adb609e28f6befc1 -size 802604 +oid sha256:81225f9e0b8b42522ba01e69e5a7207495916da59b101e694a437734ef9a3440 +size 841004 diff --git a/samples/step_0100000/04-conversational.wav b/samples/step_0100000/04-conversational.wav index a14eaddf7c8d3ed3599d6e682845c31f7b31be72..0dccfdce1e6c7583a407e3d8616d6e3dea8b3879 100644 --- a/samples/step_0100000/04-conversational.wav +++ b/samples/step_0100000/04-conversational.wav @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:b01efa57885bd9667accb51a35508d42f663a3ee5d13d292f9060c8f34d6c674 +oid sha256:9cdf15771b9bc38db03573d062ef083d2c6164f8189977ed34cca7c93e365c0a size 1129004 diff --git a/samples/step_0100000/05-long.wav b/samples/step_0100000/05-long.wav index 75091988bf2c107de641ce37e13a6902ed293bf2..2b3963243f7fd11228b290d01ae40bd6ad050707 100644 --- a/samples/step_0100000/05-long.wav +++ b/samples/step_0100000/05-long.wav @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:25713c69605a9d4727ed81d4c04d3fe6f695739f9f43a1b7eb8b86a2f0881fda -size 1699244 +oid sha256:252c18556b414370fdd7330dc92e07094b33533494e4cc87a3a6bf1a111d8524 +size 1691564 diff --git a/samples/step_0110000/01-short.wav b/samples/step_0110000/01-short.wav deleted file mode 100644 index 11ca7b5e3a26ea9b333640c45662a057899d9b5c..0000000000000000000000000000000000000000 --- a/samples/step_0110000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3c2bedd2de253a2b825fd07d15df6f1f30aec23155e0c54b0ae8428859efc735 -size 491564 diff --git a/samples/step_0110000/02-question.wav b/samples/step_0110000/02-question.wav deleted file mode 100644 index 42b780b8ccbd984e662b44573088b7516a9ab339..0000000000000000000000000000000000000000 --- a/samples/step_0110000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2a3425e8268e519b57c81703c98fbb8fe6ac02d8e12e644bdbb539bc5aa5c744 -size 675884 diff --git a/samples/step_0110000/03-numbers.wav b/samples/step_0110000/03-numbers.wav deleted file mode 100644 index 2c9ad6453b998dd16f636b04dd047e7c7671e34d..0000000000000000000000000000000000000000 --- a/samples/step_0110000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4beecb3d4d3dbd7ab99f5822e6e296503c9bb6a20a1609ad8912ba44b38a165e -size 802604 diff --git a/samples/step_0110000/04-conversational.wav b/samples/step_0110000/04-conversational.wav deleted file mode 100644 index ffbd9b7591653476b59e9258f27e8753a785c49e..0000000000000000000000000000000000000000 --- a/samples/step_0110000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:20e3a50faa40f85f34bb3f05d6948dde89ec18fe7fea4788e9054282b87758af -size 1129004 diff --git a/samples/step_0110000/05-long.wav b/samples/step_0110000/05-long.wav deleted file mode 100644 index 6b73cdb01118e6a6bcb3b8a08ee0b20ab2b50922..0000000000000000000000000000000000000000 --- a/samples/step_0110000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a3e3de209cb713ff9deee84fccdca1a3a4f0f6f8acb20f29d5a4823175c431ea -size 1699244 diff --git a/samples/step_0120000/01-short.wav b/samples/step_0120000/01-short.wav deleted file mode 100644 index da993a47d55ba1e9d51b4c57c9bed6003f54775d..0000000000000000000000000000000000000000 --- a/samples/step_0120000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3cf37bf94caac0e9446e71e23c80336049c4ca185f565fc12031e52487aab484 -size 491564 diff --git a/samples/step_0120000/02-question.wav b/samples/step_0120000/02-question.wav deleted file mode 100644 index 2ecfcd41f460261fa9b5cba406dd8fc76ce6acd6..0000000000000000000000000000000000000000 --- a/samples/step_0120000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b014a8e1972b9b73372dad88b075803971960c94fea8f787e853aa88849f5721 -size 675884 diff --git a/samples/step_0120000/03-numbers.wav b/samples/step_0120000/03-numbers.wav deleted file mode 100644 index 68e1927129ddaf740676e81384bb9d06eddf2060..0000000000000000000000000000000000000000 --- a/samples/step_0120000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:61fa694b20e10ffa176aa12690bf554d324bae123d3bd17ab18c05fb6e1d1e74 -size 802604 diff --git a/samples/step_0120000/04-conversational.wav b/samples/step_0120000/04-conversational.wav deleted file mode 100644 index 6e6add962c6fc050c3d9fb7255ed066d7ba20ce9..0000000000000000000000000000000000000000 --- a/samples/step_0120000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5b3d217630c73c741bce10dd76313a3066204b2817fd16732e3c0809f5229ade -size 1129004 diff --git a/samples/step_0120000/05-long.wav b/samples/step_0120000/05-long.wav deleted file mode 100644 index 94a26aacfacc90517dec2ee0757874529bd975eb..0000000000000000000000000000000000000000 --- a/samples/step_0120000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bb44b904b13883b7c6c1687ddac517a6eeda20f63289d9b52b27b3ceb101c070 -size 1699244 diff --git a/samples/step_0130000/01-short.wav b/samples/step_0130000/01-short.wav deleted file mode 100644 index 48877c521c2b3f3e3b823fb28d4bbf6904b16334..0000000000000000000000000000000000000000 --- a/samples/step_0130000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:411bd27078dcfa13deac7b307b3968c4ee8557a6c35b137b319a0cb12a825b91 -size 491564 diff --git a/samples/step_0130000/02-question.wav b/samples/step_0130000/02-question.wav deleted file mode 100644 index 3507550f8d3fb3c64b6f256a14e2d269165de99a..0000000000000000000000000000000000000000 --- a/samples/step_0130000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:67e0aca5ea00e34c068ec9ad34043691ffbba198ff53baad16b58303bc611640 -size 675884 diff --git a/samples/step_0130000/03-numbers.wav b/samples/step_0130000/03-numbers.wav deleted file mode 100644 index dfbb2b9e828b85c8e964cd82f3d1117763b6ba8c..0000000000000000000000000000000000000000 --- a/samples/step_0130000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f07dbeb39177293350e473d037e33a9ed5dd3d591236820e94aaa33622021535 -size 802604 diff --git a/samples/step_0130000/04-conversational.wav b/samples/step_0130000/04-conversational.wav deleted file mode 100644 index cc03ff6339310146a6745b49839bea945fad191d..0000000000000000000000000000000000000000 --- a/samples/step_0130000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b1df114f794001d27903df7375fa2264dd08a3c016b91e93d2c48506670859ea -size 1129004 diff --git a/samples/step_0130000/05-long.wav b/samples/step_0130000/05-long.wav deleted file mode 100644 index c883227c1aa3418b91a12ce29c3410edf9ad4b60..0000000000000000000000000000000000000000 --- a/samples/step_0130000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6e3bc27962a68e4f3ea94b6c5f1bbbc87954ba6aa389b3c8f8c50c2e566bd028 -size 1699244 diff --git a/samples/step_0140000/01-short.wav b/samples/step_0140000/01-short.wav deleted file mode 100644 index 7febd806b0c9580f44b76a77f8f3f5d54a4d53d9..0000000000000000000000000000000000000000 --- a/samples/step_0140000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8d0cece9d4e9c3a0f3dc1ae74c3f52b9fdfe82b4fe257f8fcc80c435293c1539 -size 491564 diff --git a/samples/step_0140000/02-question.wav b/samples/step_0140000/02-question.wav deleted file mode 100644 index 5d8f8b894c326afe27a7843fa24a22e9ed130fd1..0000000000000000000000000000000000000000 --- a/samples/step_0140000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:15234544b40592053078cf3d83d6748c2b53ea05499af74c4f84ae0343087272 -size 675884 diff --git a/samples/step_0140000/03-numbers.wav b/samples/step_0140000/03-numbers.wav deleted file mode 100644 index 08830bfc51464304724bae3a6cc0e9f0b1e4a69a..0000000000000000000000000000000000000000 --- a/samples/step_0140000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:15373a4535b07fe3b246a53c797e0adae5744a4ff44f25fed17ed1669a88e40a -size 802604 diff --git a/samples/step_0140000/04-conversational.wav b/samples/step_0140000/04-conversational.wav deleted file mode 100644 index 7f544268b3b7aac06bb6a04f77312f99e18ee8e4..0000000000000000000000000000000000000000 --- a/samples/step_0140000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:cc96b4263213a0e4c8eaf680d12d2fa4a5cda82b9e52b0ff59c2106aad122488 -size 1129004 diff --git a/samples/step_0140000/05-long.wav b/samples/step_0140000/05-long.wav deleted file mode 100644 index 3adfd406a5316966a671e71102075f3aebc908ed..0000000000000000000000000000000000000000 --- a/samples/step_0140000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a0cabfeaebca3b5e23e48f7eff8bb265425e6f74c399c8aa36a6b0045e370adb -size 1699244 diff --git a/samples/step_0150000/01-short.wav b/samples/step_0150000/01-short.wav deleted file mode 100644 index a2c16700367e2493c3d1882fabf3639581891e24..0000000000000000000000000000000000000000 --- a/samples/step_0150000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:50fb944f7762077b02aedc789fc1023ed67867470f6ed83c9a54f1a99101c9d7 -size 491564 diff --git a/samples/step_0150000/02-question.wav b/samples/step_0150000/02-question.wav deleted file mode 100644 index 2c3203549dd385dedb583bbaf71f4a58e060fd7f..0000000000000000000000000000000000000000 --- a/samples/step_0150000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5b165fbbfa5ee3365f5050bc8433a993295887d7db581e9af1768c17a8a04d68 -size 675884 diff --git a/samples/step_0150000/03-numbers.wav b/samples/step_0150000/03-numbers.wav deleted file mode 100644 index 7f45be16e5aa6b8bb4af8f3712098b9dc6f5c24f..0000000000000000000000000000000000000000 --- a/samples/step_0150000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a53533016b5f1eaef7624eb2b7cdafe9345a7a5b76759d02a91e1d3f7fbd205c -size 802604 diff --git a/samples/step_0150000/04-conversational.wav b/samples/step_0150000/04-conversational.wav deleted file mode 100644 index c4b298bb18012ac235daaba7b3cdd97f06b89909..0000000000000000000000000000000000000000 --- a/samples/step_0150000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b07d1e6acd27ef7e80653a9c753fc1a29b3e421b32de0de3ea062d85830a4366 -size 1129004 diff --git a/samples/step_0150000/05-long.wav b/samples/step_0150000/05-long.wav deleted file mode 100644 index 6bed451c55b8f2845b2eed90565257e57f427697..0000000000000000000000000000000000000000 --- a/samples/step_0150000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3e8c3818deb64fc0d7a4207c341e2b2f324ebb6896c978fd83f3453d38a675ad -size 1699244 diff --git a/samples/step_0160000/01-short.wav b/samples/step_0160000/01-short.wav deleted file mode 100644 index 431117a26170a2cbfb9f62ef0633cbc6be4f2e43..0000000000000000000000000000000000000000 --- a/samples/step_0160000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:eb05400acd9fb35898dea33ef79a96f4f3a00e83c25fe6d597372fbd1de3e388 -size 491564 diff --git a/samples/step_0160000/02-question.wav b/samples/step_0160000/02-question.wav deleted file mode 100644 index 5072b2c2a715fb0e53a881213e60836886d69327..0000000000000000000000000000000000000000 --- a/samples/step_0160000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2daedbec8ad10d2c72693c0e8159485b841a03638614ffc2ce9049202e7e0c33 -size 675884 diff --git a/samples/step_0160000/03-numbers.wav b/samples/step_0160000/03-numbers.wav deleted file mode 100644 index ffd12a18f3370938d6427f891bc21b48d40ad6ab..0000000000000000000000000000000000000000 --- a/samples/step_0160000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:72ab1423536d29f712a9be5e95b49c66e52844192e064624c2828b1757114381 -size 802604 diff --git a/samples/step_0160000/04-conversational.wav b/samples/step_0160000/04-conversational.wav deleted file mode 100644 index 4657e0be62feb7e7791dfee2e71a366e659e4b7f..0000000000000000000000000000000000000000 --- a/samples/step_0160000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b5c7e3d2d36f4f80414e0aec4b2aeb0e765a37a195479d208a9129646fb05fd5 -size 1129004 diff --git a/samples/step_0160000/05-long.wav b/samples/step_0160000/05-long.wav deleted file mode 100644 index af93355e390649cc62ac11a15561d72e6625cf66..0000000000000000000000000000000000000000 --- a/samples/step_0160000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2ca63d7fc367564dc2997841af690309298cb9fc218e8eedd1ed78996bb334cd -size 1699244 diff --git a/samples/step_0170000/01-short.wav b/samples/step_0170000/01-short.wav deleted file mode 100644 index 34a42a515178dae35addf61918710f24bdd42cb1..0000000000000000000000000000000000000000 --- a/samples/step_0170000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9d1ed8e22a999d107c5a6d9c8e83b16e91bfaca4ec5750a48133bee0a087067f -size 491564 diff --git a/samples/step_0170000/02-question.wav b/samples/step_0170000/02-question.wav deleted file mode 100644 index 9dd08ef76a5b95aa87443ec818f80cd88eb1f3c7..0000000000000000000000000000000000000000 --- a/samples/step_0170000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:96085f8499fb107fe7aeede38a7e97f58d904eb9a25988cf8f05b684336065fc -size 675884 diff --git a/samples/step_0170000/03-numbers.wav b/samples/step_0170000/03-numbers.wav deleted file mode 100644 index 8bd1bd226c23635059e79a1c35904439e2e92d1a..0000000000000000000000000000000000000000 --- a/samples/step_0170000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5574b249d390be3284b4887078354420f09883220fc6957007b2bf06ab7db6d4 -size 802604 diff --git a/samples/step_0170000/04-conversational.wav b/samples/step_0170000/04-conversational.wav deleted file mode 100644 index 695fc1fa86c5514874050e929b5a22b23e27de82..0000000000000000000000000000000000000000 --- a/samples/step_0170000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:15a2cde8d509611fa4df5e17265ba6a3c17f1d0be07954e7b1dc3f57bd2c4ae5 -size 1129004 diff --git a/samples/step_0170000/05-long.wav b/samples/step_0170000/05-long.wav deleted file mode 100644 index 0ea5765de39581a469bf631ecda6194fa60f0a49..0000000000000000000000000000000000000000 --- a/samples/step_0170000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:48f8660c932d0bc65ed6bd7e6c886d3d320f6f1de633695eed831e2c8d92b0bf -size 1699244 diff --git a/samples/step_0180000/01-short.wav b/samples/step_0180000/01-short.wav deleted file mode 100644 index 6358eca8208ed609d75d47b5bccfbeea01610059..0000000000000000000000000000000000000000 --- a/samples/step_0180000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:39706b5e94421a6758cb697a0a5c4221cf38656c867654c30230e58885993af3 -size 491564 diff --git a/samples/step_0180000/02-question.wav b/samples/step_0180000/02-question.wav deleted file mode 100644 index 4e78f7da0604f5b1614bbef48ea77a25afce2b44..0000000000000000000000000000000000000000 --- a/samples/step_0180000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d64a2cbd16018d6594677a4d1dae3342c1bde9944b69ff619eb36633292ceb72 -size 675884 diff --git a/samples/step_0180000/03-numbers.wav b/samples/step_0180000/03-numbers.wav deleted file mode 100644 index b44bbafa410935a55c563a6601989a30b46710af..0000000000000000000000000000000000000000 --- a/samples/step_0180000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a60c7dac7b1d0bcf16d1a1d8d7f7c646e04c8274f285752aedd2bcd596e37f91 -size 802604 diff --git a/samples/step_0180000/04-conversational.wav b/samples/step_0180000/04-conversational.wav deleted file mode 100644 index cdc909ea4b85f47ee106ad8101a979f4f9814d59..0000000000000000000000000000000000000000 --- a/samples/step_0180000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:cd3d1c0a165b0d786bbee8b67c8f264c4befd9a9d45942d3819eaa834c14e291 -size 1129004 diff --git a/samples/step_0180000/05-long.wav b/samples/step_0180000/05-long.wav deleted file mode 100644 index 6cb06da2c46ca1ea77f61e81eb19b5a87e353eb3..0000000000000000000000000000000000000000 --- a/samples/step_0180000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6a911c1b0ab9f28d89d62e6da2318857d7185ed4918202fe400f0702b4d6e725 -size 1699244 diff --git a/samples/step_0190000/01-short.wav b/samples/step_0190000/01-short.wav deleted file mode 100644 index 865d57f3b0f0b55dad70ebb37d2e8b7ca2b07426..0000000000000000000000000000000000000000 --- a/samples/step_0190000/01-short.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4baa5581b46bf09bc61a1c78eec171ffc69e9236c1805901aa82f06992f40ce0 -size 491564 diff --git a/samples/step_0190000/02-question.wav b/samples/step_0190000/02-question.wav deleted file mode 100644 index 035f61ef4fe19dfd876610a0518826069a3f94d1..0000000000000000000000000000000000000000 --- a/samples/step_0190000/02-question.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:451992342ad52a95b75537662d7b5c22793884aa0ce22b679df289a00597e45c -size 675884 diff --git a/samples/step_0190000/03-numbers.wav b/samples/step_0190000/03-numbers.wav deleted file mode 100644 index 0a9494a308154a13c74b4f423c2ac0f1379a140d..0000000000000000000000000000000000000000 --- a/samples/step_0190000/03-numbers.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f12139b5f53c63ef0ea0319a6ea6a901e978b8018d858100c669d3f5736137fd -size 802604 diff --git a/samples/step_0190000/04-conversational.wav b/samples/step_0190000/04-conversational.wav deleted file mode 100644 index ad162c703e5dab41591c296d7f15fa2824ba9556..0000000000000000000000000000000000000000 --- a/samples/step_0190000/04-conversational.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2b04a8d37859b9ba5ca72f3b2faf8164aa714e0da2bfa68217eb9062f7e21e13 -size 1129004 diff --git a/samples/step_0190000/05-long.wav b/samples/step_0190000/05-long.wav deleted file mode 100644 index af86629bd0ad6d697e7de256723f1505f7859a65..0000000000000000000000000000000000000000 --- a/samples/step_0190000/05-long.wav +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6a5150f41229c40cb81786b77971e054d0234d624aa69c93ca767d8ac3e3c079 -size 1699244 diff --git a/showcase.json b/showcase.json index 426d6e145a0fc9a89ce00251b44a7023c2c85d25..222a3ad0c8aa39be9573a20b02fa10e501797f14 100644 --- a/showcase.json +++ b/showcase.json @@ -1,5 +1,5 @@ { - "card_sha": "8e51b42c33d8a946", + "card_sha": "af9261b6f20dd65b", "checkpoints": { "en_base/step_0020000": { "added": "2026-10-03T02:01:41Z", @@ -42,75 +42,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T02:01:41Z", - "code_commit": "f0909e4", - "folder": "samples/step_0020000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0020000/01-short.wav", - "gen_seconds": 1.44, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0020000/02-question.wav", - "gen_seconds": 0.96, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0020000/03-numbers.wav", - "gen_seconds": 1.11, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0020000/04-conversational.wav", - "gen_seconds": 1.13, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0020000/05-long.wav", - "gen_seconds": 1.96, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.1, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00020000.pt", "step": 20000, "train_sources": [ @@ -166,75 +97,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T02:01:09Z", - "code_commit": "f0909e4", - "folder": "samples/step_0030000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0030000/01-short.wav", - "gen_seconds": 1.54, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0030000/02-question.wav", - "gen_seconds": 1.01, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0030000/03-numbers.wav", - "gen_seconds": 1.14, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0030000/04-conversational.wav", - "gen_seconds": 1.16, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 0.9911 - }, - { - "file": "samples/step_0030000/05-long.wav", - "gen_seconds": 2.17, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.1, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00030000.pt", "step": 30000, "train_sources": [ @@ -296,75 +158,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T02:00:35Z", - "code_commit": "f0909e4", - "folder": "samples/step_0040000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0040000/01-short.wav", - "gen_seconds": 2.58, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 0.9976 - }, - { - "file": "samples/step_0040000/02-question.wav", - "gen_seconds": 0.95, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0040000/03-numbers.wav", - "gen_seconds": 1.14, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0040000/04-conversational.wav", - "gen_seconds": 1.18, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0040000/05-long.wav", - "gen_seconds": 2.19, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00040000.pt", "step": 40000, "train_sources": [ @@ -420,75 +213,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T03:02:28Z", - "code_commit": "f0909e4", - "folder": "samples/step_0050000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0050000/01-short.wav", - "gen_seconds": 2.81, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0050000/02-question.wav", - "gen_seconds": 0.96, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0050000/03-numbers.wav", - "gen_seconds": 1.13, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0050000/04-conversational.wav", - "gen_seconds": 1.12, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0050000/05-long.wav", - "gen_seconds": 2.15, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00050000.pt", "step": 50000, "train_sources": [ @@ -550,75 +274,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T05:43:27Z", - "code_commit": "f0909e4", - "folder": "samples/step_0060000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0060000/01-short.wav", - "gen_seconds": 2.63, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0060000/02-question.wav", - "gen_seconds": 0.95, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0060000/03-numbers.wav", - "gen_seconds": 1.1, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0060000/04-conversational.wav", - "gen_seconds": 1.15, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0060000/05-long.wav", - "gen_seconds": 2.11, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00060000.pt", "step": 60000, "train_sources": [ @@ -674,75 +329,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T07:04:18Z", - "code_commit": "f0909e4", - "folder": "samples/step_0070000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0070000/01-short.wav", - "gen_seconds": 2.32, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0070000/02-question.wav", - "gen_seconds": 0.98, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0070000/03-numbers.wav", - "gen_seconds": 1.17, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0070000/04-conversational.wav", - "gen_seconds": 1.17, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0070000/05-long.wav", - "gen_seconds": 2.13, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00070000.pt", "step": 70000, "train_sources": [ @@ -804,75 +390,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T09:05:19Z", - "code_commit": "f0909e4", - "folder": "samples/step_0080000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0080000/01-short.wav", - "gen_seconds": 2.81, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0080000/02-question.wav", - "gen_seconds": 0.99, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0080000/03-numbers.wav", - "gen_seconds": 1.13, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0080000/04-conversational.wav", - "gen_seconds": 1.17, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0080000/05-long.wav", - "gen_seconds": 2.21, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00080000.pt", "step": 80000, "train_sources": [ @@ -928,75 +445,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T11:06:19Z", - "code_commit": "f0909e4", - "folder": "samples/step_0090000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0090000/01-short.wav", - "gen_seconds": 2.42, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0090000/02-question.wav", - "gen_seconds": 0.98, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0090000/03-numbers.wav", - "gen_seconds": 1.16, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0090000/04-conversational.wav", - "gen_seconds": 1.18, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0090000/05-long.wav", - "gen_seconds": 2.16, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00090000.pt", "step": 90000, "train_sources": [ @@ -1059,64 +507,84 @@ "recipe_arm": false, "run": "en_base", "samples": { - "added": "2026-10-03T13:07:27Z", - "code_commit": "f0909e4", + "added": "2026-10-05T01:13:14Z", + "code_commit": "029dec6", "folder": "samples/step_0100000", - "items_sha": "267be367fdfb1a46", + "items_sha": "8fcd29112a5e183c", "rows": [ { + "bandwidth_clamped": null, + "bandwidth_hz": 15568.5, "file": "samples/step_0100000/01-short.wav", - "gen_seconds": 2.63, + "gen_seconds": 2.74, "item": "01-short", + "prompt_bandwidth_hz": 15568.5, "seconds": 5.12, "watermark": "audioseal_wm_16bits", "watermark_bits": "0100110101010100", "watermark_score": 1.0 }, { + "bandwidth_clamped": "high", + "bandwidth_hz": 16839.0, "file": "samples/step_0100000/02-question.wav", - "gen_seconds": 0.97, + "gen_seconds": 0.85, "item": "02-question", - "seconds": 7.04, + "prompt_bandwidth_hz": 16968.2, + "seconds": 6.08, "watermark": "audioseal_wm_16bits", "watermark_bits": "0100110101010100", "watermark_score": 1.0 }, { + "bandwidth_clamped": null, + "bandwidth_hz": 15482.4, "file": "samples/step_0100000/03-numbers.wav", - "gen_seconds": 1.11, + "gen_seconds": 0.88, "item": "03-numbers", - "seconds": 8.36, + "prompt_bandwidth_hz": 15482.4, + "seconds": 8.76, "watermark": "audioseal_wm_16bits", "watermark_bits": "0100110101010100", "watermark_score": 1.0 }, { + "bandwidth_clamped": null, + "bandwidth_hz": 16559.0, "file": "samples/step_0100000/04-conversational.wav", - "gen_seconds": 1.12, + "gen_seconds": 0.88, "item": "04-conversational", + "prompt_bandwidth_hz": 16559.0, "seconds": 11.76, "watermark": "audioseal_wm_16bits", "watermark_bits": "0100110101010100", "watermark_score": 1.0 }, { + "bandwidth_clamped": null, + "bandwidth_hz": 15697.7, "file": "samples/step_0100000/05-long.wav", - "gen_seconds": 2.09, + "gen_seconds": 1.7, "item": "05-long", - "seconds": 17.7, + "prompt_bandwidth_hz": 15697.7, + "seconds": 17.62, "watermark": "audioseal_wm_16bits", "watermark_bits": "0100110101010100", "watermark_score": 1.0 } ], "sampler": { + "bandwidth": "auto", + "bandwidth_range_hz": [ + 9797.607421875, + 16838.96484375 + ], "cfg_mode": "joint", "cfg_rescale": 0.0, "cfg_t_max": 1.0, "cfg_t_min": 0.0, "cfg_w": 4.0, - "gpu_peak_gb": 4.08, + "gpu_peak_gb": 4.18, "noise_scale": 0.9, "out_lufs": -16.0, "schedule": "sway", @@ -1182,75 +650,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T15:48:57Z", - "code_commit": "f0909e4", - "folder": "samples/step_0110000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0110000/01-short.wav", - "gen_seconds": 2.52, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0110000/02-question.wav", - "gen_seconds": 0.99, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0110000/03-numbers.wav", - "gen_seconds": 1.13, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0110000/04-conversational.wav", - "gen_seconds": 1.17, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0110000/05-long.wav", - "gen_seconds": 2.19, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00110000.pt", "step": 110000, "train_sources": [ @@ -1312,75 +711,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T17:30:04Z", - "code_commit": "f0909e4", - "folder": "samples/step_0120000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0120000/01-short.wav", - "gen_seconds": 2.43, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0120000/02-question.wav", - "gen_seconds": 0.96, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0120000/03-numbers.wav", - "gen_seconds": 1.12, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0120000/04-conversational.wav", - "gen_seconds": 1.13, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0120000/05-long.wav", - "gen_seconds": 2.13, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00120000.pt", "step": 120000, "train_sources": [ @@ -1436,75 +766,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T19:31:41Z", - "code_commit": "f0909e4", - "folder": "samples/step_0130000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0130000/01-short.wav", - "gen_seconds": 3.05, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0130000/02-question.wav", - "gen_seconds": 0.97, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0130000/03-numbers.wav", - "gen_seconds": 1.17, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0130000/04-conversational.wav", - "gen_seconds": 1.09, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0130000/05-long.wav", - "gen_seconds": 2.3, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00130000.pt", "step": 130000, "train_sources": [ @@ -1566,75 +827,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-03T21:33:08Z", - "code_commit": "f0909e4", - "folder": "samples/step_0140000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0140000/01-short.wav", - "gen_seconds": 2.71, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0140000/02-question.wav", - "gen_seconds": 0.96, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0140000/03-numbers.wav", - "gen_seconds": 1.15, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0140000/04-conversational.wav", - "gen_seconds": 1.13, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0140000/05-long.wav", - "gen_seconds": 2.14, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00140000.pt", "step": 140000, "train_sources": [ @@ -1656,109 +848,40 @@ } }, "en_base/step_0150000": { - "added": "2026-10-04T00:14:14Z", - "catalog": "echo_en_v1", - "catalog_clips": 3400911, - "catalog_hours": 9543.28, - "catalog_voices": 3587, - "code_commit": "f0909e4", - "codec_id": "Aratako/Semantic-DACVAE-Japanese@737fbad2a679f880", - "data_version": "data-4255c6734a7d5b5e", - "files": { - "config.json": { - "bytes": 5714, - "sha256": "a37b6985556773fc548ff7697516c084046732c58bb8e05a7f99da621ed5f644" - }, - "model.safetensors": { - "bytes": 825673000, - "sha256": "e30d138d30eda5517f6fa45d2a60c9d18024c844741299854c0adc66c9fa3a44" - } - }, - "folder": "checkpoints/step_0150000", - "lineage": [], - "metrics": { - "echo_dev": { - "n": 260, - "sim_o": 0.7937824732982195, - "utmos": 3.927360293498406, - "wer": 0.005294659300184162 - } - }, - "note": "Main model (M4 base pretraining): the 206 M base preset on the full provided data", - "params_m": 206.41, - "preset": "base", - "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", - "recipe_arm": false, - "run": "en_base", - "samples": { - "added": "2026-10-04T00:14:14Z", - "code_commit": "f0909e4", - "folder": "samples/step_0150000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0150000/01-short.wav", - "gen_seconds": 2.36, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0150000/02-question.wav", - "gen_seconds": 1.0, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0150000/03-numbers.wav", - "gen_seconds": 1.15, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0150000/04-conversational.wav", - "gen_seconds": 1.17, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0150000/05-long.wav", - "gen_seconds": 2.12, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 + "added": "2026-10-04T00:14:14Z", + "catalog": "echo_en_v1", + "catalog_clips": 3400911, + "catalog_hours": 9543.28, + "catalog_voices": 3587, + "code_commit": "f0909e4", + "codec_id": "Aratako/Semantic-DACVAE-Japanese@737fbad2a679f880", + "data_version": "data-4255c6734a7d5b5e", + "files": { + "config.json": { + "bytes": 5714, + "sha256": "a37b6985556773fc548ff7697516c084046732c58bb8e05a7f99da621ed5f644" + }, + "model.safetensors": { + "bytes": 825673000, + "sha256": "e30d138d30eda5517f6fa45d2a60c9d18024c844741299854c0adc66c9fa3a44" } }, + "folder": "checkpoints/step_0150000", + "lineage": [], + "metrics": { + "echo_dev": { + "n": 260, + "sim_o": 0.7937824732982195, + "utmos": 3.927360293498406, + "wer": 0.005294659300184162 + } + }, + "note": "Main model (M4 base pretraining): the 206 M base preset on the full provided data", + "params_m": 206.41, + "preset": "base", + "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", + "recipe_arm": false, + "run": "en_base", "source": "en_base/export/model_step_00150000.pt", "step": 150000, "train_sources": [ @@ -1820,75 +943,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-04T01:55:12Z", - "code_commit": "f0909e4", - "folder": "samples/step_0160000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0160000/01-short.wav", - "gen_seconds": 3.22, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0160000/02-question.wav", - "gen_seconds": 0.98, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0160000/03-numbers.wav", - "gen_seconds": 1.15, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0160000/04-conversational.wav", - "gen_seconds": 1.17, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0160000/05-long.wav", - "gen_seconds": 2.25, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00160000.pt", "step": 160000, "train_sources": [ @@ -1944,75 +998,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-04T03:56:10Z", - "code_commit": "f0909e4", - "folder": "samples/step_0170000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0170000/01-short.wav", - "gen_seconds": 2.51, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0170000/02-question.wav", - "gen_seconds": 1.0, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0170000/03-numbers.wav", - "gen_seconds": 1.15, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0170000/04-conversational.wav", - "gen_seconds": 1.16, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0170000/05-long.wav", - "gen_seconds": 2.19, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00170000.pt", "step": 170000, "train_sources": [ @@ -2074,75 +1059,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-04T05:57:11Z", - "code_commit": "f0909e4", - "folder": "samples/step_0180000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0180000/01-short.wav", - "gen_seconds": 2.82, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0180000/02-question.wav", - "gen_seconds": 0.97, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0180000/03-numbers.wav", - "gen_seconds": 1.11, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0180000/04-conversational.wav", - "gen_seconds": 1.16, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0180000/05-long.wav", - "gen_seconds": 2.13, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00180000.pt", "step": 180000, "train_sources": [ @@ -2198,75 +1114,6 @@ "provenance": "catalog echo_en_v1 (data-4255c6734a7d5b5e): echo_en", "recipe_arm": false, "run": "en_base", - "samples": { - "added": "2026-10-04T08:18:19Z", - "code_commit": "f0909e4", - "folder": "samples/step_0190000", - "items_sha": "267be367fdfb1a46", - "rows": [ - { - "file": "samples/step_0190000/01-short.wav", - "gen_seconds": 2.29, - "item": "01-short", - "seconds": 5.12, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0190000/02-question.wav", - "gen_seconds": 0.98, - "item": "02-question", - "seconds": 7.04, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0190000/03-numbers.wav", - "gen_seconds": 1.16, - "item": "03-numbers", - "seconds": 8.36, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0190000/04-conversational.wav", - "gen_seconds": 1.17, - "item": "04-conversational", - "seconds": 11.76, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - }, - { - "file": "samples/step_0190000/05-long.wav", - "gen_seconds": 2.19, - "item": "05-long", - "seconds": 17.7, - "watermark": "audioseal_wm_16bits", - "watermark_bits": "0100110101010100", - "watermark_score": 1.0 - } - ], - "sampler": { - "cfg_mode": "joint", - "cfg_rescale": 0.0, - "cfg_t_max": 1.0, - "cfg_t_min": 0.0, - "cfg_w": 4.0, - "gpu_peak_gb": 4.08, - "noise_scale": 0.9, - "out_lufs": -16.0, - "schedule": "sway", - "source": "eval_args.json", - "steps": 32, - "sway": -1.0, - "w_speaker": 3.5, - "w_text": 2.5 - } - }, "source": "en_base/export/model_step_00190000.pt", "step": 190000, "train_sources": [ @@ -2288,7 +1135,83 @@ } } }, - "format": "mytts-showcase/1", + "featured": { + "bandwidth": "auto", + "bandwidth_range_from": "p5..p95 of the training catalog echo_en_v1 (data-4255c6734a7d5b5e, 3,397,565 train clips), as evaluate.py --bandwidth auto recorded it", + "bandwidth_range_hz": [ + 9797.607421875, + 16838.96484375 + ], + "compare": { + "bandwidth": "default", + "checkpoints": [ + { + "echo_dev_v2": { + "n": 1000, + "share_sim_o_below_0_2": 0.0, + "sim_o": 0.816516, + "utmos": 4.00165, + "wer": 0.004919 + }, + "seed_dev": { + "n": 1090, + "share_sim_o_below_0_2": 0.0569, + "sim_o": 0.499364, + "utmos": 3.78412, + "wer": 0.015374 + }, + "step": 100000 + }, + { + "echo_dev_v2": { + "n": 1000, + "share_sim_o_below_0_2": 0.001, + "sim_o": 0.820022, + "utmos": 4.137699, + "wer": 0.004575 + }, + "seed_dev": { + "n": 1090, + "share_sim_o_below_0_2": 0.5147, + "sim_o": 0.222872, + "utmos": 3.004047, + "wer": 0.020555 + }, + "step": 190000 + } + ] + }, + "decided": "2026-10-04: owner decision, the M5 base is the step-100k EMA export with --bandwidth auto", + "file": "en_release_featured.json", + "issue": 177, + "key": "en_base/step_0100000", + "run": "en_base", + "scores": { + "echo_dev_v2": { + "n": 1000, + "sentences": 1000, + "share_sim_o_below_0_2": 0.0, + "sim_o": 0.816373, + "takes": 1, + "utmos": 3.9925, + "voices": 198, + "wer": 0.005264 + }, + "seed_dev": { + "n": 1090, + "sentences": 545, + "share_sim_o_below_0_2": 0.0239, + "sim_o": 0.526291, + "takes": 2, + "utmos": 3.804776, + "wer": 0.014205 + } + }, + "step": 100000, + "stopped_at": 194271, + "stopped_on": "2026-10-04" + }, + "format": "mytts-showcase/2", "items": [ { "category": "short", @@ -2309,30 +1232,30 @@ "category": "question", "item": "02-question", "pitch": "lower", - "prompt_clip": "train-01111-342", - "prompt_f0_hz": 148.6, + "prompt_clip": "train-00024-33", + "prompt_f0_hz": 119.3, "prompt_license": "apache-2.0 (SynDataLab-EN/echo-clones-4m-en, held-out dev voice)", - "prompt_seconds": 4.319, - "prompt_sha256": "fb9d2e89e2480bd3ff7615fe09db9bb49bdd61c44cd1381098450980f5a0e694", - "prompt_text": "Wait, that mole— has it always looked like that? No, stop.", + "prompt_seconds": 5.712, + "prompt_sha256": "ec5010cbb72864e62cd95a87840c6f74a750f4ac135ddcda43ce34417b63c346", + "prompt_text": "Look, I'm not saying it's easy, but you always pull it together. Just take it step by step, you know?", "seed": 0, "text": "Have you ever noticed that the quietest person in the room usually has the most interesting story to tell?", - "voice": "spk_0342", + "voice": "spk_0327", "voice_set": "echo" }, { "category": "numbers", "item": "03-numbers", "pitch": "higher", - "prompt_clip": "train-09741-169", - "prompt_f0_hz": 180.8, + "prompt_clip": "train-02023-409", + "prompt_f0_hz": 200.7, "prompt_license": "apache-2.0 (SynDataLab-EN/echo-clones-4m-en, held-out dev voice)", "prompt_seconds": 7.198, - "prompt_sha256": "887e3d57be68bdf2ec8f869e753ba95155abfd1d8667a6d91419cbd7283ca391", - "prompt_text": "this woman on the train was literally eating a whole bag of chips, like, crunching so loud, and I'm sitting there like, okay, do I move? whatever.", + "prompt_sha256": "1d667de2ea2cc8713b440ee979664f94031b4148333a5a61d59de5946d0de008", + "prompt_text": "Would you just— I know you mean well, but every time you mention it I feel like an idiot. It's probably nothing.", "seed": 0, "text": "Dr. Patel moved my appointment to Tuesday, March 3rd, at 4:15 p.m., and the co-pay went up from $20 to $35.", - "voice": "spk_2669", + "voice": "spk_0409", "voice_set": "echo" }, { @@ -2354,31 +1277,213 @@ "category": "long", "item": "05-long", "pitch": "lowest", - "prompt_clip": "train-00254-493", - "prompt_f0_hz": 108.2, + "prompt_clip": "train-09627-464", + "prompt_f0_hz": 102.1, "prompt_license": "apache-2.0 (SynDataLab-EN/echo-clones-4m-en, held-out dev voice)", - "prompt_seconds": 7.059, - "prompt_sha256": "b0047e2d525a230a7d05e70ac838100b92c425687e7cc0a3c5ab0e065c04af95", - "prompt_text": "Honestly I think we need to just call IT and have them... you know, actually fix it this time. I'm too old for this.", + "prompt_seconds": 8.034, + "prompt_sha256": "8311aeaf5ac99a1b940d1945c3390de11f845c02328163fe4fadad8e7943c4e0", + "prompt_text": "I mean, come on, it's not like I woke up and decided to have the worst day ever. Things just went wrong one after another, like dominoes!", "seed": 0, "text": "When the storm passed, the whole neighborhood came outside to look at the damage, and although a few fences had fallen and the old oak tree had lost its biggest branch, everyone was relieved that nobody was hurt. They spent the rest of the afternoon clearing the street and sharing whatever food they had left.", - "voice": "spk_1437", + "voice": "spk_1964", "voice_set": "echo" } ], - "items_sha": "267be367fdfb1a46", + "items_sha": "8fcd29112a5e183c", "kind": "release", "policy": { "final_only": true, "main_min_step": 20000, "min_step": 20000 }, + "prompt_choice": { + "added": "2026-10-05T01:13:20Z", + "code_commit": "029dec6", + "evidence": { + "prompt_share": 0.629, + "prompts": 45, + "text_share": 0.042, + "texts": 10 + }, + "fingerprint": "9618a7549aa69b0f", + "issue": 177, + "items_sha": "8fcd29112a5e183c", + "key": "en_base/step_0100000", + "rows": [ + { + "bandwidth_clamped": "low", + "bandwidth_hz": 9797.6, + "before": { + "category": "question", + "item": "02-question", + "pitch": "lower", + "prompt_clip": "train-01111-342", + "prompt_f0_hz": 148.6, + "prompt_license": "apache-2.0 (SynDataLab-EN/echo-clones-4m-en, held-out dev voice)", + "prompt_seconds": 4.319, + "prompt_sha256": "fb9d2e89e2480bd3ff7615fe09db9bb49bdd61c44cd1381098450980f5a0e694", + "prompt_text": "Wait, that mole— has it always looked like that? No, stop.", + "seed": 0, + "text": "Have you ever noticed that the quietest person in the room usually has the most interesting story to tell?", + "voice": "spk_0342", + "voice_set": "echo" + }, + "file": "samples/prompt_choice/old_output/02-question.wav", + "gen_seconds": 0.88, + "item": "02-question", + "measured": { + "bandwidth": "auto", + "new": { + "output_band_hz": 15023.4, + "output_dnsmos_ovrl": 3.425, + "output_utmos": 4.248, + "prompt_band_hz": 16968.2 + }, + "old": { + "output_band_hz": 8226.6, + "output_dnsmos_ovrl": 3.291, + "output_utmos": 3.844, + "prompt_band_hz": 8850.1 + }, + "seed": 0, + "step": 100000 + }, + "problem": "a dull recording (band 8.9 kHz)", + "prompt_bandwidth_hz": 8850.1, + "prompt_file": "samples/prompt_choice/old_prompt/02-question.wav", + "prompt_sha256": "337e2af17f4096f6afa7beb9a151859157c3ee8ec3cbfeca9e63aa2a61529cab", + "seconds": 7.04, + "watermark": "audioseal_wm_16bits", + "watermark_bits": "0100110101010100", + "watermark_score": 1.0, + "why": "the prompt's band is 8.9 kHz (below the training range's p5, 9.8 kHz, so bandwidth auto clamps it there): the step-100k output of 02-question is band-limited at 8.2 kHz and scores UTMOS 3.84." + }, + { + "bandwidth_clamped": "high", + "bandwidth_hz": 16839.0, + "before": { + "category": "numbers", + "item": "03-numbers", + "pitch": "higher", + "prompt_clip": "train-09741-169", + "prompt_f0_hz": 180.8, + "prompt_license": "apache-2.0 (SynDataLab-EN/echo-clones-4m-en, held-out dev voice)", + "prompt_seconds": 7.198, + "prompt_sha256": "887e3d57be68bdf2ec8f869e753ba95155abfd1d8667a6d91419cbd7283ca391", + "prompt_text": "this woman on the train was literally eating a whole bag of chips, like, crunching so loud, and I'm sitting there like, okay, do I move? whatever.", + "seed": 0, + "text": "Dr. Patel moved my appointment to Tuesday, March 3rd, at 4:15 p.m., and the co-pay went up from $20 to $35.", + "voice": "spk_2669", + "voice_set": "echo" + }, + "file": "samples/prompt_choice/old_output/03-numbers.wav", + "gen_seconds": 0.88, + "item": "03-numbers", + "measured": { + "bandwidth": "auto", + "new": { + "output_band_hz": 15304.7, + "output_dnsmos_ovrl": 3.552, + "output_utmos": 4.385, + "prompt_band_hz": 15482.4 + }, + "old": { + "output_band_hz": 17460.9, + "output_dnsmos_ovrl": 3.399, + "output_utmos": 3.698, + "prompt_band_hz": 17743.4 + }, + "seed": 0, + "step": 100000 + }, + "problem": "background noise (signal-to-noise 38 dB)", + "prompt_bandwidth_hz": 17743.4, + "prompt_file": "samples/prompt_choice/old_prompt/03-numbers.wav", + "prompt_sha256": "c7949d82713134b3b4480cfe85cd599e5ddfc73e860a96de1b09bee04b9102fd", + "seconds": 8.36, + "watermark": "audioseal_wm_16bits", + "watermark_bits": "0100110101010100", + "watermark_score": 1.0, + "why": "a noise bed: the quiet frames' 8-16 kHz level is -31 dB (the rule: -45 dB or lower), signal-to-noise 38 dB, prompt UTMOS 4.14 (the sweep's lowest third). Its odd 17.7 kHz band is that noise; the step-100k output of 03-numbers copies it (17.5 kHz) and scores UTMOS 3.70." + }, + { + "bandwidth_clamped": null, + "bandwidth_hz": 10938.9, + "before": { + "category": "long", + "item": "05-long", + "pitch": "lowest", + "prompt_clip": "train-00254-493", + "prompt_f0_hz": 108.2, + "prompt_license": "apache-2.0 (SynDataLab-EN/echo-clones-4m-en, held-out dev voice)", + "prompt_seconds": 7.059, + "prompt_sha256": "b0047e2d525a230a7d05e70ac838100b92c425687e7cc0a3c5ab0e065c04af95", + "prompt_text": "Honestly I think we need to just call IT and have them... you know, actually fix it this time. I'm too old for this.", + "seed": 0, + "text": "When the storm passed, the whole neighborhood came outside to look at the damage, and although a few fences had fallen and the old oak tree had lost its biggest branch, everyone was relieved that nobody was hurt. They spent the rest of the afternoon clearing the street and sharing whatever food they had left.", + "voice": "spk_1437", + "voice_set": "echo" + }, + "file": "samples/prompt_choice/old_output/05-long.wav", + "gen_seconds": 1.7, + "item": "05-long", + "measured": { + "bandwidth": "auto", + "new": { + "output_band_hz": 14460.9, + "output_dnsmos_ovrl": 3.393, + "output_utmos": 4.384, + "prompt_band_hz": 15697.7 + }, + "old": { + "output_band_hz": 8273.4, + "output_dnsmos_ovrl": 3.37, + "output_utmos": 4.149, + "prompt_band_hz": 10938.9 + }, + "seed": 0, + "step": 100000 + }, + "problem": "a dull recording (band 10.9 kHz)", + "prompt_bandwidth_hz": 10938.9, + "prompt_file": "samples/prompt_choice/old_prompt/05-long.wav", + "prompt_sha256": "b0047e2d525a230a7d05e70ac838100b92c425687e7cc0a3c5ab0e065c04af95", + "seconds": 17.7, + "watermark": "audioseal_wm_16bits", + "watermark_bits": "0100110101010100", + "watermark_score": 1.0, + "why": "the prompt's band is 10.9 kHz: the step-100k output of 05-long is band-limited at 8.3 kHz and scores UTMOS 4.15." + } + ], + "sampler": { + "bandwidth": "auto", + "bandwidth_range_hz": [ + 9797.607421875, + 16838.96484375 + ], + "cfg_mode": "joint", + "cfg_rescale": 0.0, + "cfg_t_max": 1.0, + "cfg_t_min": 0.0, + "cfg_w": 4.0, + "gpu_peak_gb": 3.76, + "noise_scale": 0.9, + "out_lufs": -16.0, + "schedule": "sway", + "source": "eval_args.json", + "steps": 32, + "sway": -1.0, + "w_speaker": 3.5, + "w_text": 2.5 + }, + "step": 100000 + }, "prompts": { "01-short": "5d1822e59071fe22d6910d08f4c2c49735853be3cf50916b1e7cef17c246f477", - "02-question": "337e2af17f4096f6afa7beb9a151859157c3ee8ec3cbfeca9e63aa2a61529cab", - "03-numbers": "c7949d82713134b3b4480cfe85cd599e5ddfc73e860a96de1b09bee04b9102fd", + "02-question": "ec5010cbb72864e62cd95a87840c6f74a750f4ac135ddcda43ce34417b63c346", + "03-numbers": "4f1d94263fbe54bdcd239706933b97365dac4d34ea62f75e319da06821cdeff7", "04-conversational": "b7c93eb3bc3eb63828440322c7c79ec38df309d6476c5e6c7403b73576d4b031", - "05-long": "b0047e2d525a230a7d05e70ac838100b92c425687e7cc0a3c5ab0e065c04af95" + "05-long": "9f6966023f8946ea8d9d915dbd3a8ac00c242217b131dcc979eed1063df2b9e0" }, "recipe": { "data.catalog": "catalogs/echo_en_v1",