diff --git a/README.md b/README.md index 2e87cb0fd10bc83a648f66d53c5f098eb876e350..4db28afa9528f68415538245009b70427e84e4f0 100644 --- a/README.md +++ b/README.md @@ -1,15 +1,77 @@ --- -title: Loudkit -emoji: 📚 -colorFrom: pink -colorTo: green +title: loudkit +emoji: 🔊 +colorFrom: gray +colorTo: red sdk: gradio -sdk_version: 6.26.0 -python_version: '3.12' +sdk_version: 5.50.0 +python_version: "3.12.12" app_file: app.py pinned: false license: apache-2.0 -short_description: showcase of the loudkit TTS library +short_description: On-device TTS. Twenty voices, ten languages, one engine. +models: + - loudreader/loudr-1 +preload_from_hub: + - loudreader/loudr-1 loudr-1.safetensors,loudr-1-enrollment.safetensors,ve.safetensors,manifest.json,release.json,SHA256SUMS,voices/carmen.safetensors,voices/colette.safetensors,voices/dante.safetensors,voices/darkman.safetensors,voices/dave.safetensors,voices/freja.safetensors,voices/gosia.safetensors,voices/henri.safetensors,voices/ines.safetensors,voices/joe.safetensors,voices/kathleen.safetensors,voices/kerstin.safetensors,voices/nathalie.safetensors,voices/nils.safetensors,voices/paola.safetensors,voices/pim.safetensors,voices/selma.safetensors,voices/soren.safetensors,voices/thorsten.safetensors,voices/tugao.safetensors --- -Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference +# loudkit + +Twenty voices across ten languages, from [loudreader/loudr-1](https://huggingface.co/loudreader/loudr-1), +running the [loudkit](https://github.com/loudreader/loudkit) engine on ZeroGPU. + +## Three tabs + +- **Listen.** Twenty voices, each beside the reference recording it was enrolled + from. These files were rendered ahead of time and ship in this repo. This tab + uses no GPU. +- **Speak.** Your text, in any of the twenty voices. Up to 1,000 characters. +- **Clone.** Your own voice, from about ten seconds of audio. + +ZeroGPU bills GPU time to the visitor, not to the owner. An anonymous visitor +gets about two minutes a day. A signed-in free account gets about five. Listening +costs none of it. + +## Cloning and consent + +Clone your own voice, or a voice you have permission to use. + +- The microphone is the default path. +- An upload is secondary, and needs an explicit confirmation. +- Recordings are deleted when the request ends. Nothing is kept. + +See [RESPONSIBLE_USE](https://github.com/loudreader/loudkit/blob/main/RESPONSIBLE_USE.md). + +## Determinism + +The Speak tab has a determinism check. It renders the same text twice at the same +seed and prints the SHA-256 of both waveforms. They match. + +That holds within this build and this device. loudkit promises a bit-identical +waveform for the same seed, build, backend and input. It does not promise that +your machine matches this GPU. See the +[identity contract](https://github.com/loudreader/loudkit/blob/main/docs/reference/IDENTITY-CONTRACT.md). + +## Run it locally + +```bash +pip install "loudkit[torch,audio,enroll,hub]" +``` + +```python +import loudkit as lk + +engine = lk.load("loudreader/loudr-1") +voice = lk.voice("kathleen", repo="loudreader/loudr-1") +engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav") +``` + +Audio in this Space, and from `Result.save`, carries C2PA provenance: the +algorithm fingerprint, the recipe and the seed. + +## Voice sources + +Every voice is enrolled from a public-domain or openly licensed recording. +`voices.json` in this repo carries the full record for each one: donor, source, +licence, consent, and the SHA-256 of both the reference and the sample. diff --git a/app.py b/app.py new file mode 100644 index 0000000000000000000000000000000000000000..3e2e44e07163d6b6d66f366ce80678012a99f0f9 --- /dev/null +++ b/app.py @@ -0,0 +1,541 @@ +"""The loudkit demo Space: hear twenty voices, speak your own text, clone your own. + +ZeroGPU bills GPU time to the *visitor*, not to the owner: an anonymous visitor +gets about two minutes a day, a signed-in free account about five. A demo whose +first click spends that budget is one most people bounce off before they have +heard anything at all. So the Listen tab is twenty pre-rendered files served +straight out of this repo — no GPU, no queue, no quota — and the GPU is spent +only on what a visitor types or records. + +The engine and the enroller are both built on `cuda` at module level, which is +what ZeroGPU asks for: CUDA transfers are optimised for start-up placement, and +lazy-loading inside a `@spaces.GPU` function is explicitly discouraged. Each +decorated call then runs in a freshly forked, short-lived process, which is also +why there is no `torch.compile` and no CUDA graph capture here: both pay their +cost once per process and would never amortise. + +Cloning is exposed, which the CPU scaffold this replaces deliberately did not do. +The reasoning that kept it out was about consent, not about capability, so the +consent is built into the shape of the tab rather than written beside it: the +microphone is the default path, an upload is secondary and gated on an explicit +confirmation, and neither recording outlives the request that carried it. +""" + +from __future__ import annotations + +import contextlib +import dataclasses +import hashlib +import json +import os +import tempfile +from pathlib import Path + +# Before torch, and before anything that imports torch. The module installs the +# CUDA emulation that lets a module-level `.to("cuda")` succeed on a machine +# that has no GPU attached yet. +import spaces + +import gradio as gr +import numpy as np + +import loudkit as lk +from loudkit.backends.torch_backend import build_torch_enroller +from loudkit.hub import resolve_enrollment_checkpoint, resolve_voice_encoder + +REPO = "loudreader/loudr-1" +DEVICE = "cuda" +HERE = Path(__file__).parent + +# The CPU scaffold capped text at 300 characters because CPU synthesis ran at +# roughly a tenth of real time. On a GPU the cap is about the visitor's daily +# quota instead, which is a far looser bound: 1000 characters is ~70 s of speech. +MAX_CHARS = 1_000 +MAX_CLONE_CHARS = 400 +MAX_PROBE_CHARS = 200 + +# The enroller refuses anything over 30 s and wants 5 to 10. The prompt is built +# from the first 10 s; the speaker embedding reads whatever else is there, so a +# little past the prompt window is useful and 20 s stays clear of the refusal. +ENROLL_SECONDS = 20.0 + +DOCS = "https://github.com/loudreader/loudkit" +IDENTITY_CONTRACT = f"{DOCS}/blob/main/docs/reference/IDENTITY-CONTRACT.md" +RESPONSIBLE_USE = f"{DOCS}/blob/main/RESPONSIBLE_USE.md" + +ROSTER = json.loads((HERE / "voices.json").read_text(encoding="utf-8")) +BY_NAME = {entry["name"]: entry for entry in ROSTER} +ORDERED = sorted(ROSTER, key=lambda e: (e["language"], e["name"])) +VOICE_CHOICES = [(f"{e['name']} · {e['language']} ({e['gender']})", e["name"]) for e in ORDERED] + +# -------------------------------------------------------------------------- +# Module-level model placement, per the ZeroGPU contract. +# -------------------------------------------------------------------------- + +engine = lk.load(REPO, device=DEVICE) + +# Voice profiles are numpy, not torch, so they are device-agnostic and cost a +# few hundred kilobytes each. Loading all twenty up front means switching voice +# in the Speak tab never blocks on a download. +PROFILES = {entry["name"]: lk.voice(entry["name"], repo=REPO) for entry in ROSTER} + +# Enrollment reads the other half of the release: the speech tokenizer and the +# speaker encoder, which synthesis never touches, plus the utterance voice +# encoder that sits beside both. `lk.enroll()` builds this per call by design; +# a Space would pay the load on every clone, so it is built once here instead. +enroller = build_torch_enroller( + str(resolve_enrollment_checkpoint(REPO)), + device=DEVICE, + voice_encoder_weights=str(resolve_voice_encoder(REPO)), +) + +FINGERPRINT = engine.algorithm.fingerprint() + +_LANGUAGE_NAMES = {e["language_id"]: e["language"] for e in ROSTER} +LANGUAGE_CHOICES = [("Follow the voice", "")] + [ + (f"{_LANGUAGE_NAMES.get(code, code)} ({code})", code) for code in lk.languages() +] + +# -------------------------------------------------------------------------- +# Helpers +# -------------------------------------------------------------------------- + + +def _sha256_audio(audio: np.ndarray) -> str: + """Hash the waveform, not the file. + + `Result.save` appends a C2PA manifest carrying a wall-clock creation time, + which the library itself calls the one byte range in which two identical + renders may legitimately differ. Hashing the saved WAV would therefore print + two different digests for two identical renders and read as a determinism + failure. The waveform is what the identity contract makes its promise about, + so the waveform is what gets hashed. + """ + return hashlib.sha256(np.ascontiguousarray(audio, dtype=np.float32).tobytes()).hexdigest() + + +def _write(result: lk.Result, *, voice: str, language: str) -> str: + out = tempfile.NamedTemporaryFile(suffix=".wav", delete=False) + out.close() + # Provenance on: the manifest carries the fingerprint, the recipe and the + # seed, which is the machine-readable marking a synthetic-speech demo should + # be handing out by default. + result.save(out.name, voice=voice, language=language) + return out.name + + +def _stats(result: lk.Result) -> str: + seconds = len(result.audio) / result.sample_rate + return ( + f"**{seconds:.1f} s of audio.** {result.timings.describe(seconds)}\n\n" + f"`algo[{result.algorithm_fingerprint}]` · seed `{result.seed}` · " + f"speed `{result.speed:g}x` · {result.sample_rate} Hz" + ) + + +def _estimate(text: str, *, passes: int = 1, overhead: float = 15.0) -> int: + """Seconds of GPU to ask for. + + Speech runs at roughly 14 characters a second, and the render is asked to + keep up with better than real time; the overhead covers the process fork and + the first real CUDA touch. Asking for too much costs queue priority but not + quota, which is charged on effective duration, so this leans generous. + """ + audio_seconds = len((text or "").strip()) / 14.0 + return int(min(180.0, overhead + passes * max(4.0, audio_seconds * 0.9))) + + +def _check(text: str, limit: int) -> str: + text = (text or "").strip() + if not text: + raise gr.Error("Type something to say.") + if len(text) > limit: + raise gr.Error(f"Keep it under {limit:,} characters here. The library itself takes 10,000.") + return text + + +# -------------------------------------------------------------------------- +# Listen. No GPU: these files were rendered ahead of time and ship in the repo. +# -------------------------------------------------------------------------- + + +def listen(name: str): + entry = BY_NAME[name] + sample, reference, source = entry["sample"], entry["reference"], entry["source"] + + lines = [ + f"### {entry['name']}. {entry['language']} ({entry['gender']}).", + "", + f"> {sample['text']}", + "", + f"From *{sample['work']}*, seed `{sample['seed']}`.", + "", + f"- Reference recording: {reference['duration_s']:.1f} s, {reference['construction']}.", + f"- Source: [{source['name']}]({source['url']}), {source['license']}.", + f"- Consent: {source['consent']}.", + ] + similarity = entry.get("speaker_similarity") + if similarity is not None: + lines.append(f"- Speaker similarity to the reference: {similarity:.3f}.") + lines.append(f"- Voice profile: `{entry['profile']['hf_path']}`.") + + return ( + str(HERE / sample["audio"]), + str(HERE / reference["public_preview"]), + "\n".join(lines), + ) + + +ROSTER_TABLE = [ + [ + entry["name"], + entry["language"], + entry["gender"], + entry["source"]["license"], + f"{entry['speaker_similarity']:.3f}" if entry.get("speaker_similarity") is not None else "", + ] + for entry in ORDERED +] + + +# -------------------------------------------------------------------------- +# Speak. GPU. +# -------------------------------------------------------------------------- + + +def _speak_duration(text, name, language, seed, speed): + return _estimate(text, overhead=15.0) + + +@spaces.GPU(duration=_speak_duration) +def speak(text: str, name: str, language: str, seed: float, speed: float): + text = _check(text, MAX_CHARS) + result = engine.synthesize_long( + text, + PROFILES[name], + seed=int(seed), + language=language or None, + speed=float(speed), + ) + label = language or BY_NAME[name]["language_id"] + return _write(result, voice=name, language=label), _stats(result) + + +# -------------------------------------------------------------------------- +# Clone. GPU. The microphone is the default path; an upload is gated. +# -------------------------------------------------------------------------- + + +def _clone_duration(mic, upload, consent, text, language, seed, speed): + # Enrollment is a fixed cost on top of the render: two encoders and a + # tokenizer over at most 20 s of audio. + return _estimate(text, overhead=30.0) + + +@spaces.GPU(duration=_clone_duration) +def clone(mic, upload, consent: bool, text: str, language: str, seed: float, speed: float): + source = mic or upload + if not source: + raise gr.Error("Record yourself first, or upload a clip you are allowed to use.") + if upload and not mic and not consent: + raise gr.Error("Confirm the uploaded voice is yours, or that you have permission to use it.") + text = _check(text, MAX_CLONE_CHARS) + + try: + import librosa + + samples, _ = librosa.load(source, sr=24_000, mono=True) + limit = int(ENROLL_SECONDS * 24_000) + if samples.size > limit: + samples = samples[:limit] + + try: + profile = enroller.enroll(samples, 24_000, name="your voice") + except ValueError as exc: + # The library's own messages name the bound and describe a good + # input, which is more useful than anything restated here. + raise gr.Error(str(exc)) from exc + + # `enroll` writes no language, so every cloned voice would claim English + # and read its text through the English funnel. + profile = dataclasses.replace(profile, language=language or "en") + + result = engine.synthesize_long( + text, profile, seed=int(seed), language=language or None, speed=float(speed) + ) + return _write(result, voice="cloned", language=profile.language), _stats(result) + finally: + # Nothing the visitor recorded outlives the request that carried it. + with contextlib.suppress(OSError): + os.unlink(source) + + +# -------------------------------------------------------------------------- +# Determinism probe. GPU. Renders the same text twice at the same seed. +# -------------------------------------------------------------------------- + + +def _probe_duration(text, name, seed): + return _estimate(text, passes=2, overhead=20.0) + + +@spaces.GPU(duration=_probe_duration) +def probe(text: str, name: str, seed: float): + text = _check(text, MAX_PROBE_CHARS) + profile = PROFILES[name] + first = engine.synthesize_long(text, profile, seed=int(seed)) + second = engine.synthesize_long(text, profile, seed=int(seed)) + + left, right = _sha256_audio(first.audio), _sha256_audio(second.audio) + verdict = "Identical." if left == right else "Different. Please report this." + + return "\n".join( + [ + f"**{verdict}**", + "", + "```", + f"render 1 sha256 {left}", + f"render 2 sha256 {right}", + f" algo[{first.algorithm_fingerprint}] seed {int(seed)}", + "```", + "", + "Identical within this build and this device. loudkit promises a " + "bit-identical waveform for the same seed, build, backend and input. " + "It does not promise that your laptop matches this GPU. " + f"[Read the identity contract]({IDENTITY_CONTRACT}).", + ] + ) + + +# -------------------------------------------------------------------------- +# Interface +# -------------------------------------------------------------------------- + +# loudreader.io: cream ground, ink text, black pill buttons at 14px. +CSS = """ +#lk-head h1 { font-size: 2.15rem; margin-bottom: .25rem; letter-spacing: -.02em; } +#lk-head p { margin-top: 0; } +.lk-pill { + display: inline-block; padding: .2rem .75rem; margin: .15rem .35rem .15rem 0; + border: 1px solid #ded8ce; border-radius: 999px; font-size: .8rem; + color: #374151; background: #fffdfa; +} +.lk-card { background: #fffdfa; border: 1px solid #e7e1d7; border-radius: 14px; padding: .35rem 1rem; } +footer { display: none !important; } +""" + +# Gradio follows the visitor's system theme unless told otherwise, and this +# palette is light-first. Without this the ink-on-cream tokens below land under +# a dark stylesheet and the text turns near-white on a cream ground. +FORCE_LIGHT = """ +() => { + const url = new URL(window.location); + if (url.searchParams.get('__theme') !== 'light') { + url.searchParams.set('__theme', 'light'); + window.location.replace(url.href); + } +} +""" + +THEME = gr.themes.Soft( + primary_hue=gr.themes.colors.gray, + neutral_hue=gr.themes.colors.stone, + font=[gr.themes.GoogleFont("Inter"), "system-ui", "sans-serif"], +).set( + body_background_fill="#f7f5f2", + body_text_color="#111827", + body_text_color_subdued="#4b5563", + block_background_fill="#fffdfa", + block_border_color="#e7e1d7", + border_color_primary="#e7e1d7", + input_background_fill="#ffffff", + button_primary_background_fill="#111827", + button_primary_background_fill_hover="#374151", + button_primary_text_color="#ffffff", + button_large_radius="14px", + button_small_radius="14px", +) + +with gr.Blocks(title="loudkit", theme=THEME, css=CSS, js=FORCE_LIGHT, fill_width=False) as demo: + gr.Markdown( + f""" +# Twenty voices. Ten languages. One engine. + +On-device text to speech, running here on ZeroGPU. +[Model](https://huggingface.co/{REPO}) · [Code]({DOCS}) · [Responsible use]({RESPONSIBLE_USE}) + +Listening costs no GPU +Speaking and cloning spend your daily quota +algo[{FINGERPRINT}] +""", + elem_id="lk-head", + ) + + with gr.Tabs(): + # ---------------- Listen ---------------- + with gr.Tab("Listen"): + gr.Markdown( + "Twenty voices, rendered ahead of time and served as files. " + "This tab uses no GPU and spends none of your quota. " + "Each voice is paired with the reference recording it was enrolled from." + ) + with gr.Row(): + with gr.Column(scale=1): + pick = gr.Dropdown( + VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True + ) + made = gr.Audio(label="loudkit", type="filepath", interactive=False) + ref = gr.Audio(label="Reference recording", type="filepath", interactive=False) + with gr.Column(scale=1): + card = gr.Markdown(elem_classes="lk-card") + + with gr.Accordion("The whole roster", open=False): + gr.Dataframe( + value=ROSTER_TABLE, + headers=["Voice", "Language", "Gender", "Licence", "Similarity"], + interactive=False, + wrap=True, + ) + + pick.change(listen, pick, [made, ref, card]) + demo.load(listen, pick, [made, ref, card]) + + # ---------------- Speak ---------------- + with gr.Tab("Speak"): + gr.Markdown( + f"Your text, in one of the twenty voices. " + f"Up to {MAX_CHARS:,} characters here. The library itself takes 10,000. " + "This tab spends your ZeroGPU quota." + ) + with gr.Row(): + with gr.Column(scale=3): + say = gr.Textbox( + label="Text", + placeholder="Hello from loudkit.", + lines=4, + max_length=MAX_CHARS, + ) + with gr.Column(scale=2): + say_voice = gr.Dropdown( + VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True + ) + say_lang = gr.Dropdown( + LANGUAGE_CHOICES, value="", label="Read the text as" + ) + with gr.Row(): + say_seed = gr.Number(value=7, precision=0, label="Seed") + say_speed = gr.Slider( + lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed" + ) + say_go = gr.Button("Speak", variant="primary") + say_out = gr.Audio(label="Speech", type="filepath") + say_stats = gr.Markdown() + + say_go.click( + speak, + [say, say_voice, say_lang, say_seed, say_speed], + [say_out, say_stats], + ) + + with gr.Accordion("Determinism check", open=False): + gr.Markdown( + "This renders the same text twice at the same seed and hashes " + "both waveforms. The digests must match." + ) + with gr.Row(): + probe_text = gr.Textbox( + value="The same seed gives the same audio.", + label="Text", + max_length=MAX_PROBE_CHARS, + scale=3, + ) + probe_voice = gr.Dropdown( + VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", scale=2 + ) + probe_seed = gr.Number(value=7, precision=0, label="Seed", scale=1) + probe_go = gr.Button("Render twice") + probe_out = gr.Markdown() + probe_go.click(probe, [probe_text, probe_voice, probe_seed], probe_out) + + # ---------------- Clone ---------------- + with gr.Tab("Clone"): + gr.Markdown( + f""" +Clone a voice from a short recording, then speak with it. + +- Record 5 to 10 seconds. Read anything. Speak normally. +- Clone only your own voice, or a voice you have permission to use. +- Nothing you record is kept. The recording is deleted when the request ends. +- See [Responsible use]({RESPONSIBLE_USE}). +""" + ) + with gr.Row(): + with gr.Column(scale=1): + mic = gr.Audio( + sources=["microphone"], + type="filepath", + label="Record yourself", + ) + with gr.Accordion("Upload a file instead", open=False): + upload = gr.Audio( + sources=["upload"], type="filepath", label="Audio file" + ) + consent = gr.Checkbox( + value=False, + label=( + "This is my own voice, or I have permission from the " + "person who owns it." + ), + ) + with gr.Column(scale=1): + clone_text = gr.Textbox( + label="Text to speak", + placeholder="Now in my own voice.", + lines=3, + max_length=MAX_CLONE_CHARS, + ) + clone_lang = gr.Dropdown( + LANGUAGE_CHOICES[1:], value="en", label="Language of the text" + ) + with gr.Row(): + clone_seed = gr.Number(value=7, precision=0, label="Seed") + clone_speed = gr.Slider( + lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed" + ) + clone_go = gr.Button("Clone and speak", variant="primary") + clone_out = gr.Audio(label="Speech", type="filepath") + clone_stats = gr.Markdown() + + clone_go.click( + clone, + [mic, upload, consent, clone_text, clone_lang, clone_seed, clone_speed], + [clone_out, clone_stats], + ) + + gr.Markdown( + f""" +--- +Run the same engine locally, where nothing is queued and nothing is metered. + +```bash +pip install "loudkit[torch,audio,enroll,hub]" +``` + +```python +import loudkit as lk + +engine = lk.load("{REPO}") +voice = lk.voice("kathleen", repo="{REPO}") +engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav") +``` + +Output files carry C2PA provenance: the fingerprint, the recipe and the seed. +""" + ) + +# The engine holds one set of weights and renders with an internal producer +# thread. One render at a time keeps two requests off the same buffers. +demo.queue(default_concurrency_limit=1, max_size=24) + +if __name__ == "__main__": + demo.launch() diff --git a/audio/carmen.opus b/audio/carmen.opus new file mode 100644 index 0000000000000000000000000000000000000000..e1e18609163ff3cd67315d85497818f104d33a58 Binary files /dev/null and b/audio/carmen.opus differ diff --git a/audio/colette.opus b/audio/colette.opus new file mode 100644 index 0000000000000000000000000000000000000000..682389d876f822ca755d906f05db376a18ff68d9 Binary files /dev/null and b/audio/colette.opus differ diff --git a/audio/dante.opus b/audio/dante.opus new file mode 100644 index 0000000000000000000000000000000000000000..6628f6858d4f71d379c47411baee9507f8192928 Binary files /dev/null and b/audio/dante.opus differ diff --git a/audio/darkman.opus b/audio/darkman.opus new file mode 100644 index 0000000000000000000000000000000000000000..7a6ce2e7c75e4206475b532df2fe8a9850a29929 Binary files /dev/null and b/audio/darkman.opus differ diff --git a/audio/dave.opus b/audio/dave.opus new file mode 100644 index 0000000000000000000000000000000000000000..502d2da2e6fdf140c6310a3336457eb210a7bc31 Binary files /dev/null and b/audio/dave.opus differ diff --git a/audio/freja.opus b/audio/freja.opus new file mode 100644 index 0000000000000000000000000000000000000000..2160a36de61672adf91284b9c74c692b59f07206 Binary files /dev/null and b/audio/freja.opus differ diff --git a/audio/gosia.opus b/audio/gosia.opus new file mode 100644 index 0000000000000000000000000000000000000000..853a342df5107a4894770834ef16a8539ecf1f6f Binary files /dev/null and b/audio/gosia.opus differ diff --git a/audio/henri.opus b/audio/henri.opus new file mode 100644 index 0000000000000000000000000000000000000000..97c0f1a5a9bd18e1516d97a70f26fe3a5b3ae0bb Binary files /dev/null and b/audio/henri.opus differ diff --git a/audio/ines.opus b/audio/ines.opus new file mode 100644 index 0000000000000000000000000000000000000000..2f988d7adb9dc0bd7041089ac470d0cbee02c832 Binary files /dev/null and b/audio/ines.opus differ diff --git a/audio/joe.opus b/audio/joe.opus new file mode 100644 index 0000000000000000000000000000000000000000..d35624d4b0cd92eb44d278737e92a65c95db9892 Binary files /dev/null and b/audio/joe.opus differ diff --git a/audio/kathleen.opus b/audio/kathleen.opus new file mode 100644 index 0000000000000000000000000000000000000000..afab998420c69de122c666484f135471301633ef Binary files /dev/null and b/audio/kathleen.opus differ diff --git a/audio/kerstin.opus b/audio/kerstin.opus new file mode 100644 index 0000000000000000000000000000000000000000..bc9b99ead23656b1e5f2ca7af21146bc1923b6ef Binary files /dev/null and b/audio/kerstin.opus differ diff --git a/audio/nathalie.opus b/audio/nathalie.opus new file mode 100644 index 0000000000000000000000000000000000000000..78d74b015d59796e31d798558ffe6c846d4e5619 Binary files /dev/null and b/audio/nathalie.opus differ diff --git a/audio/nils.opus b/audio/nils.opus new file mode 100644 index 0000000000000000000000000000000000000000..9f317ca8b1aba3b9aa3dd0e86719cda67125b6d8 Binary files /dev/null and b/audio/nils.opus differ diff --git a/audio/paola.opus b/audio/paola.opus new file mode 100644 index 0000000000000000000000000000000000000000..6f936ee395156750e75bbb580eb239a62b32eb3a Binary files /dev/null and b/audio/paola.opus differ diff --git a/audio/pim.opus b/audio/pim.opus new file mode 100644 index 0000000000000000000000000000000000000000..5d31d74acd05d2cc5a8b191d3c78f06387e1eb05 Binary files /dev/null and b/audio/pim.opus differ diff --git a/audio/refs/carmen.opus b/audio/refs/carmen.opus new file mode 100644 index 0000000000000000000000000000000000000000..220d1a403a3b535d284ce5c4008e3a5fcb736945 Binary files /dev/null and b/audio/refs/carmen.opus differ diff --git a/audio/refs/colette.opus b/audio/refs/colette.opus new file mode 100644 index 0000000000000000000000000000000000000000..dce18d4f09e5b07e10f9c46441fca118b4d17b67 Binary files /dev/null and b/audio/refs/colette.opus differ diff --git a/audio/refs/dante.opus b/audio/refs/dante.opus new file mode 100644 index 0000000000000000000000000000000000000000..b58368afe0db03b2d2f960c75d91280146d5f901 Binary files /dev/null and b/audio/refs/dante.opus differ diff --git a/audio/refs/darkman.opus b/audio/refs/darkman.opus new file mode 100644 index 0000000000000000000000000000000000000000..cb98a474936dbbdb79b3020552e955242f61ef0d Binary files /dev/null and b/audio/refs/darkman.opus differ diff --git a/audio/refs/dave.opus b/audio/refs/dave.opus new file mode 100644 index 0000000000000000000000000000000000000000..ca3c66e199e7bd900d1d4180aa91b65aaba5d9e8 Binary files /dev/null and b/audio/refs/dave.opus differ diff --git a/audio/refs/freja.opus b/audio/refs/freja.opus new file mode 100644 index 0000000000000000000000000000000000000000..adbb18927dc20821a83eb11190e9faa2aa50efb3 Binary files /dev/null and b/audio/refs/freja.opus differ diff --git a/audio/refs/gosia.opus b/audio/refs/gosia.opus new file mode 100644 index 0000000000000000000000000000000000000000..ac8e283877021b29edb5bd8e27c296973cae3c07 Binary files /dev/null and b/audio/refs/gosia.opus differ diff --git a/audio/refs/henri.opus b/audio/refs/henri.opus new file mode 100644 index 0000000000000000000000000000000000000000..59d6873d132f1a4b61081a27cb1817fa1d463bbe Binary files /dev/null and b/audio/refs/henri.opus differ diff --git a/audio/refs/ines.opus b/audio/refs/ines.opus new file mode 100644 index 0000000000000000000000000000000000000000..b31012129731dabfcdd0829dabb54a8e09d5914f Binary files /dev/null and b/audio/refs/ines.opus differ diff --git a/audio/refs/joe.opus b/audio/refs/joe.opus new file mode 100644 index 0000000000000000000000000000000000000000..732465cbc480985e84fe73887ca79a520a33e374 Binary files /dev/null and b/audio/refs/joe.opus differ diff --git a/audio/refs/kathleen.opus b/audio/refs/kathleen.opus new file mode 100644 index 0000000000000000000000000000000000000000..03fa0d9136a7614ef66b4d9627042992e6904be2 Binary files /dev/null and b/audio/refs/kathleen.opus differ diff --git a/audio/refs/kerstin.opus b/audio/refs/kerstin.opus new file mode 100644 index 0000000000000000000000000000000000000000..d6ca625df4b5c103120b206c35af9cc2c7b47936 Binary files /dev/null and b/audio/refs/kerstin.opus differ diff --git a/audio/refs/nathalie.opus b/audio/refs/nathalie.opus new file mode 100644 index 0000000000000000000000000000000000000000..5bc6b148d91e82fa1caa46ce33993fd078319262 Binary files /dev/null and b/audio/refs/nathalie.opus differ diff --git a/audio/refs/nils.opus b/audio/refs/nils.opus new file mode 100644 index 0000000000000000000000000000000000000000..bdfc6564eeeb9a75cf8f03f202369a2ae0604380 Binary files /dev/null and b/audio/refs/nils.opus differ diff --git a/audio/refs/paola.opus b/audio/refs/paola.opus new file mode 100644 index 0000000000000000000000000000000000000000..cef273a388c73c7e2e3f370ce4af34b56c954344 Binary files /dev/null and b/audio/refs/paola.opus differ diff --git a/audio/refs/pim.opus b/audio/refs/pim.opus new file mode 100644 index 0000000000000000000000000000000000000000..df2d2719e8e384f380a958142f3d700406ce6e61 Binary files /dev/null and b/audio/refs/pim.opus differ diff --git a/audio/refs/selma.opus b/audio/refs/selma.opus new file mode 100644 index 0000000000000000000000000000000000000000..d6bfaab9d0c0da5e7fe91eb4d2e799e388c21c89 Binary files /dev/null and b/audio/refs/selma.opus differ diff --git a/audio/refs/soren.opus b/audio/refs/soren.opus new file mode 100644 index 0000000000000000000000000000000000000000..dda8de236b964dbe5fd8b28e1311cb7fe65540d8 Binary files /dev/null and b/audio/refs/soren.opus differ diff --git a/audio/refs/thorsten.opus b/audio/refs/thorsten.opus new file mode 100644 index 0000000000000000000000000000000000000000..5e89b94fbd091c56363285eeebc1fe81deedccac Binary files /dev/null and b/audio/refs/thorsten.opus differ diff --git a/audio/refs/tugao.opus b/audio/refs/tugao.opus new file mode 100644 index 0000000000000000000000000000000000000000..3df5f5af70b92a74f4a006e1a75581f82d027328 Binary files /dev/null and b/audio/refs/tugao.opus differ diff --git a/audio/selma.opus b/audio/selma.opus new file mode 100644 index 0000000000000000000000000000000000000000..635b80f10f7832625903ac39d8bf0e831df43eb8 Binary files /dev/null and b/audio/selma.opus differ diff --git a/audio/soren.opus b/audio/soren.opus new file mode 100644 index 0000000000000000000000000000000000000000..2ed33a17736947e4ed72c1b4c5fc05b93cc548a9 Binary files /dev/null and b/audio/soren.opus differ diff --git a/audio/thorsten.opus b/audio/thorsten.opus new file mode 100644 index 0000000000000000000000000000000000000000..278cbffa5785f98db6cd5fa1394f1eff47e6f5ef Binary files /dev/null and b/audio/thorsten.opus differ diff --git a/audio/tugao.opus b/audio/tugao.opus new file mode 100644 index 0000000000000000000000000000000000000000..71f242f67742e53eecfaeb6363a4779696d3d48c Binary files /dev/null and b/audio/tugao.opus differ diff --git a/hf-loudkit/.gitattributes b/hf-loudkit/.gitattributes new file mode 100644 index 0000000000000000000000000000000000000000..a6344aac8c09253b3b630fb776ae94478aa0275b --- /dev/null +++ b/hf-loudkit/.gitattributes @@ -0,0 +1,35 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text diff --git a/hf-loudkit/README.md b/hf-loudkit/README.md new file mode 100644 index 0000000000000000000000000000000000000000..4db28afa9528f68415538245009b70427e84e4f0 --- /dev/null +++ b/hf-loudkit/README.md @@ -0,0 +1,77 @@ +--- +title: loudkit +emoji: 🔊 +colorFrom: gray +colorTo: red +sdk: gradio +sdk_version: 5.50.0 +python_version: "3.12.12" +app_file: app.py +pinned: false +license: apache-2.0 +short_description: On-device TTS. Twenty voices, ten languages, one engine. +models: + - loudreader/loudr-1 +preload_from_hub: + - loudreader/loudr-1 loudr-1.safetensors,loudr-1-enrollment.safetensors,ve.safetensors,manifest.json,release.json,SHA256SUMS,voices/carmen.safetensors,voices/colette.safetensors,voices/dante.safetensors,voices/darkman.safetensors,voices/dave.safetensors,voices/freja.safetensors,voices/gosia.safetensors,voices/henri.safetensors,voices/ines.safetensors,voices/joe.safetensors,voices/kathleen.safetensors,voices/kerstin.safetensors,voices/nathalie.safetensors,voices/nils.safetensors,voices/paola.safetensors,voices/pim.safetensors,voices/selma.safetensors,voices/soren.safetensors,voices/thorsten.safetensors,voices/tugao.safetensors +--- + +# loudkit + +Twenty voices across ten languages, from [loudreader/loudr-1](https://huggingface.co/loudreader/loudr-1), +running the [loudkit](https://github.com/loudreader/loudkit) engine on ZeroGPU. + +## Three tabs + +- **Listen.** Twenty voices, each beside the reference recording it was enrolled + from. These files were rendered ahead of time and ship in this repo. This tab + uses no GPU. +- **Speak.** Your text, in any of the twenty voices. Up to 1,000 characters. +- **Clone.** Your own voice, from about ten seconds of audio. + +ZeroGPU bills GPU time to the visitor, not to the owner. An anonymous visitor +gets about two minutes a day. A signed-in free account gets about five. Listening +costs none of it. + +## Cloning and consent + +Clone your own voice, or a voice you have permission to use. + +- The microphone is the default path. +- An upload is secondary, and needs an explicit confirmation. +- Recordings are deleted when the request ends. Nothing is kept. + +See [RESPONSIBLE_USE](https://github.com/loudreader/loudkit/blob/main/RESPONSIBLE_USE.md). + +## Determinism + +The Speak tab has a determinism check. It renders the same text twice at the same +seed and prints the SHA-256 of both waveforms. They match. + +That holds within this build and this device. loudkit promises a bit-identical +waveform for the same seed, build, backend and input. It does not promise that +your machine matches this GPU. See the +[identity contract](https://github.com/loudreader/loudkit/blob/main/docs/reference/IDENTITY-CONTRACT.md). + +## Run it locally + +```bash +pip install "loudkit[torch,audio,enroll,hub]" +``` + +```python +import loudkit as lk + +engine = lk.load("loudreader/loudr-1") +voice = lk.voice("kathleen", repo="loudreader/loudr-1") +engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav") +``` + +Audio in this Space, and from `Result.save`, carries C2PA provenance: the +algorithm fingerprint, the recipe and the seed. + +## Voice sources + +Every voice is enrolled from a public-domain or openly licensed recording. +`voices.json` in this repo carries the full record for each one: donor, source, +licence, consent, and the SHA-256 of both the reference and the sample. diff --git a/hf-loudkit/app.py b/hf-loudkit/app.py new file mode 100644 index 0000000000000000000000000000000000000000..3e2e44e07163d6b6d66f366ce80678012a99f0f9 --- /dev/null +++ b/hf-loudkit/app.py @@ -0,0 +1,541 @@ +"""The loudkit demo Space: hear twenty voices, speak your own text, clone your own. + +ZeroGPU bills GPU time to the *visitor*, not to the owner: an anonymous visitor +gets about two minutes a day, a signed-in free account about five. A demo whose +first click spends that budget is one most people bounce off before they have +heard anything at all. So the Listen tab is twenty pre-rendered files served +straight out of this repo — no GPU, no queue, no quota — and the GPU is spent +only on what a visitor types or records. + +The engine and the enroller are both built on `cuda` at module level, which is +what ZeroGPU asks for: CUDA transfers are optimised for start-up placement, and +lazy-loading inside a `@spaces.GPU` function is explicitly discouraged. Each +decorated call then runs in a freshly forked, short-lived process, which is also +why there is no `torch.compile` and no CUDA graph capture here: both pay their +cost once per process and would never amortise. + +Cloning is exposed, which the CPU scaffold this replaces deliberately did not do. +The reasoning that kept it out was about consent, not about capability, so the +consent is built into the shape of the tab rather than written beside it: the +microphone is the default path, an upload is secondary and gated on an explicit +confirmation, and neither recording outlives the request that carried it. +""" + +from __future__ import annotations + +import contextlib +import dataclasses +import hashlib +import json +import os +import tempfile +from pathlib import Path + +# Before torch, and before anything that imports torch. The module installs the +# CUDA emulation that lets a module-level `.to("cuda")` succeed on a machine +# that has no GPU attached yet. +import spaces + +import gradio as gr +import numpy as np + +import loudkit as lk +from loudkit.backends.torch_backend import build_torch_enroller +from loudkit.hub import resolve_enrollment_checkpoint, resolve_voice_encoder + +REPO = "loudreader/loudr-1" +DEVICE = "cuda" +HERE = Path(__file__).parent + +# The CPU scaffold capped text at 300 characters because CPU synthesis ran at +# roughly a tenth of real time. On a GPU the cap is about the visitor's daily +# quota instead, which is a far looser bound: 1000 characters is ~70 s of speech. +MAX_CHARS = 1_000 +MAX_CLONE_CHARS = 400 +MAX_PROBE_CHARS = 200 + +# The enroller refuses anything over 30 s and wants 5 to 10. The prompt is built +# from the first 10 s; the speaker embedding reads whatever else is there, so a +# little past the prompt window is useful and 20 s stays clear of the refusal. +ENROLL_SECONDS = 20.0 + +DOCS = "https://github.com/loudreader/loudkit" +IDENTITY_CONTRACT = f"{DOCS}/blob/main/docs/reference/IDENTITY-CONTRACT.md" +RESPONSIBLE_USE = f"{DOCS}/blob/main/RESPONSIBLE_USE.md" + +ROSTER = json.loads((HERE / "voices.json").read_text(encoding="utf-8")) +BY_NAME = {entry["name"]: entry for entry in ROSTER} +ORDERED = sorted(ROSTER, key=lambda e: (e["language"], e["name"])) +VOICE_CHOICES = [(f"{e['name']} · {e['language']} ({e['gender']})", e["name"]) for e in ORDERED] + +# -------------------------------------------------------------------------- +# Module-level model placement, per the ZeroGPU contract. +# -------------------------------------------------------------------------- + +engine = lk.load(REPO, device=DEVICE) + +# Voice profiles are numpy, not torch, so they are device-agnostic and cost a +# few hundred kilobytes each. Loading all twenty up front means switching voice +# in the Speak tab never blocks on a download. +PROFILES = {entry["name"]: lk.voice(entry["name"], repo=REPO) for entry in ROSTER} + +# Enrollment reads the other half of the release: the speech tokenizer and the +# speaker encoder, which synthesis never touches, plus the utterance voice +# encoder that sits beside both. `lk.enroll()` builds this per call by design; +# a Space would pay the load on every clone, so it is built once here instead. +enroller = build_torch_enroller( + str(resolve_enrollment_checkpoint(REPO)), + device=DEVICE, + voice_encoder_weights=str(resolve_voice_encoder(REPO)), +) + +FINGERPRINT = engine.algorithm.fingerprint() + +_LANGUAGE_NAMES = {e["language_id"]: e["language"] for e in ROSTER} +LANGUAGE_CHOICES = [("Follow the voice", "")] + [ + (f"{_LANGUAGE_NAMES.get(code, code)} ({code})", code) for code in lk.languages() +] + +# -------------------------------------------------------------------------- +# Helpers +# -------------------------------------------------------------------------- + + +def _sha256_audio(audio: np.ndarray) -> str: + """Hash the waveform, not the file. + + `Result.save` appends a C2PA manifest carrying a wall-clock creation time, + which the library itself calls the one byte range in which two identical + renders may legitimately differ. Hashing the saved WAV would therefore print + two different digests for two identical renders and read as a determinism + failure. The waveform is what the identity contract makes its promise about, + so the waveform is what gets hashed. + """ + return hashlib.sha256(np.ascontiguousarray(audio, dtype=np.float32).tobytes()).hexdigest() + + +def _write(result: lk.Result, *, voice: str, language: str) -> str: + out = tempfile.NamedTemporaryFile(suffix=".wav", delete=False) + out.close() + # Provenance on: the manifest carries the fingerprint, the recipe and the + # seed, which is the machine-readable marking a synthetic-speech demo should + # be handing out by default. + result.save(out.name, voice=voice, language=language) + return out.name + + +def _stats(result: lk.Result) -> str: + seconds = len(result.audio) / result.sample_rate + return ( + f"**{seconds:.1f} s of audio.** {result.timings.describe(seconds)}\n\n" + f"`algo[{result.algorithm_fingerprint}]` · seed `{result.seed}` · " + f"speed `{result.speed:g}x` · {result.sample_rate} Hz" + ) + + +def _estimate(text: str, *, passes: int = 1, overhead: float = 15.0) -> int: + """Seconds of GPU to ask for. + + Speech runs at roughly 14 characters a second, and the render is asked to + keep up with better than real time; the overhead covers the process fork and + the first real CUDA touch. Asking for too much costs queue priority but not + quota, which is charged on effective duration, so this leans generous. + """ + audio_seconds = len((text or "").strip()) / 14.0 + return int(min(180.0, overhead + passes * max(4.0, audio_seconds * 0.9))) + + +def _check(text: str, limit: int) -> str: + text = (text or "").strip() + if not text: + raise gr.Error("Type something to say.") + if len(text) > limit: + raise gr.Error(f"Keep it under {limit:,} characters here. The library itself takes 10,000.") + return text + + +# -------------------------------------------------------------------------- +# Listen. No GPU: these files were rendered ahead of time and ship in the repo. +# -------------------------------------------------------------------------- + + +def listen(name: str): + entry = BY_NAME[name] + sample, reference, source = entry["sample"], entry["reference"], entry["source"] + + lines = [ + f"### {entry['name']}. {entry['language']} ({entry['gender']}).", + "", + f"> {sample['text']}", + "", + f"From *{sample['work']}*, seed `{sample['seed']}`.", + "", + f"- Reference recording: {reference['duration_s']:.1f} s, {reference['construction']}.", + f"- Source: [{source['name']}]({source['url']}), {source['license']}.", + f"- Consent: {source['consent']}.", + ] + similarity = entry.get("speaker_similarity") + if similarity is not None: + lines.append(f"- Speaker similarity to the reference: {similarity:.3f}.") + lines.append(f"- Voice profile: `{entry['profile']['hf_path']}`.") + + return ( + str(HERE / sample["audio"]), + str(HERE / reference["public_preview"]), + "\n".join(lines), + ) + + +ROSTER_TABLE = [ + [ + entry["name"], + entry["language"], + entry["gender"], + entry["source"]["license"], + f"{entry['speaker_similarity']:.3f}" if entry.get("speaker_similarity") is not None else "", + ] + for entry in ORDERED +] + + +# -------------------------------------------------------------------------- +# Speak. GPU. +# -------------------------------------------------------------------------- + + +def _speak_duration(text, name, language, seed, speed): + return _estimate(text, overhead=15.0) + + +@spaces.GPU(duration=_speak_duration) +def speak(text: str, name: str, language: str, seed: float, speed: float): + text = _check(text, MAX_CHARS) + result = engine.synthesize_long( + text, + PROFILES[name], + seed=int(seed), + language=language or None, + speed=float(speed), + ) + label = language or BY_NAME[name]["language_id"] + return _write(result, voice=name, language=label), _stats(result) + + +# -------------------------------------------------------------------------- +# Clone. GPU. The microphone is the default path; an upload is gated. +# -------------------------------------------------------------------------- + + +def _clone_duration(mic, upload, consent, text, language, seed, speed): + # Enrollment is a fixed cost on top of the render: two encoders and a + # tokenizer over at most 20 s of audio. + return _estimate(text, overhead=30.0) + + +@spaces.GPU(duration=_clone_duration) +def clone(mic, upload, consent: bool, text: str, language: str, seed: float, speed: float): + source = mic or upload + if not source: + raise gr.Error("Record yourself first, or upload a clip you are allowed to use.") + if upload and not mic and not consent: + raise gr.Error("Confirm the uploaded voice is yours, or that you have permission to use it.") + text = _check(text, MAX_CLONE_CHARS) + + try: + import librosa + + samples, _ = librosa.load(source, sr=24_000, mono=True) + limit = int(ENROLL_SECONDS * 24_000) + if samples.size > limit: + samples = samples[:limit] + + try: + profile = enroller.enroll(samples, 24_000, name="your voice") + except ValueError as exc: + # The library's own messages name the bound and describe a good + # input, which is more useful than anything restated here. + raise gr.Error(str(exc)) from exc + + # `enroll` writes no language, so every cloned voice would claim English + # and read its text through the English funnel. + profile = dataclasses.replace(profile, language=language or "en") + + result = engine.synthesize_long( + text, profile, seed=int(seed), language=language or None, speed=float(speed) + ) + return _write(result, voice="cloned", language=profile.language), _stats(result) + finally: + # Nothing the visitor recorded outlives the request that carried it. + with contextlib.suppress(OSError): + os.unlink(source) + + +# -------------------------------------------------------------------------- +# Determinism probe. GPU. Renders the same text twice at the same seed. +# -------------------------------------------------------------------------- + + +def _probe_duration(text, name, seed): + return _estimate(text, passes=2, overhead=20.0) + + +@spaces.GPU(duration=_probe_duration) +def probe(text: str, name: str, seed: float): + text = _check(text, MAX_PROBE_CHARS) + profile = PROFILES[name] + first = engine.synthesize_long(text, profile, seed=int(seed)) + second = engine.synthesize_long(text, profile, seed=int(seed)) + + left, right = _sha256_audio(first.audio), _sha256_audio(second.audio) + verdict = "Identical." if left == right else "Different. Please report this." + + return "\n".join( + [ + f"**{verdict}**", + "", + "```", + f"render 1 sha256 {left}", + f"render 2 sha256 {right}", + f" algo[{first.algorithm_fingerprint}] seed {int(seed)}", + "```", + "", + "Identical within this build and this device. loudkit promises a " + "bit-identical waveform for the same seed, build, backend and input. " + "It does not promise that your laptop matches this GPU. " + f"[Read the identity contract]({IDENTITY_CONTRACT}).", + ] + ) + + +# -------------------------------------------------------------------------- +# Interface +# -------------------------------------------------------------------------- + +# loudreader.io: cream ground, ink text, black pill buttons at 14px. +CSS = """ +#lk-head h1 { font-size: 2.15rem; margin-bottom: .25rem; letter-spacing: -.02em; } +#lk-head p { margin-top: 0; } +.lk-pill { + display: inline-block; padding: .2rem .75rem; margin: .15rem .35rem .15rem 0; + border: 1px solid #ded8ce; border-radius: 999px; font-size: .8rem; + color: #374151; background: #fffdfa; +} +.lk-card { background: #fffdfa; border: 1px solid #e7e1d7; border-radius: 14px; padding: .35rem 1rem; } +footer { display: none !important; } +""" + +# Gradio follows the visitor's system theme unless told otherwise, and this +# palette is light-first. Without this the ink-on-cream tokens below land under +# a dark stylesheet and the text turns near-white on a cream ground. +FORCE_LIGHT = """ +() => { + const url = new URL(window.location); + if (url.searchParams.get('__theme') !== 'light') { + url.searchParams.set('__theme', 'light'); + window.location.replace(url.href); + } +} +""" + +THEME = gr.themes.Soft( + primary_hue=gr.themes.colors.gray, + neutral_hue=gr.themes.colors.stone, + font=[gr.themes.GoogleFont("Inter"), "system-ui", "sans-serif"], +).set( + body_background_fill="#f7f5f2", + body_text_color="#111827", + body_text_color_subdued="#4b5563", + block_background_fill="#fffdfa", + block_border_color="#e7e1d7", + border_color_primary="#e7e1d7", + input_background_fill="#ffffff", + button_primary_background_fill="#111827", + button_primary_background_fill_hover="#374151", + button_primary_text_color="#ffffff", + button_large_radius="14px", + button_small_radius="14px", +) + +with gr.Blocks(title="loudkit", theme=THEME, css=CSS, js=FORCE_LIGHT, fill_width=False) as demo: + gr.Markdown( + f""" +# Twenty voices. Ten languages. One engine. + +On-device text to speech, running here on ZeroGPU. +[Model](https://huggingface.co/{REPO}) · [Code]({DOCS}) · [Responsible use]({RESPONSIBLE_USE}) + +Listening costs no GPU +Speaking and cloning spend your daily quota +algo[{FINGERPRINT}] +""", + elem_id="lk-head", + ) + + with gr.Tabs(): + # ---------------- Listen ---------------- + with gr.Tab("Listen"): + gr.Markdown( + "Twenty voices, rendered ahead of time and served as files. " + "This tab uses no GPU and spends none of your quota. " + "Each voice is paired with the reference recording it was enrolled from." + ) + with gr.Row(): + with gr.Column(scale=1): + pick = gr.Dropdown( + VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True + ) + made = gr.Audio(label="loudkit", type="filepath", interactive=False) + ref = gr.Audio(label="Reference recording", type="filepath", interactive=False) + with gr.Column(scale=1): + card = gr.Markdown(elem_classes="lk-card") + + with gr.Accordion("The whole roster", open=False): + gr.Dataframe( + value=ROSTER_TABLE, + headers=["Voice", "Language", "Gender", "Licence", "Similarity"], + interactive=False, + wrap=True, + ) + + pick.change(listen, pick, [made, ref, card]) + demo.load(listen, pick, [made, ref, card]) + + # ---------------- Speak ---------------- + with gr.Tab("Speak"): + gr.Markdown( + f"Your text, in one of the twenty voices. " + f"Up to {MAX_CHARS:,} characters here. The library itself takes 10,000. " + "This tab spends your ZeroGPU quota." + ) + with gr.Row(): + with gr.Column(scale=3): + say = gr.Textbox( + label="Text", + placeholder="Hello from loudkit.", + lines=4, + max_length=MAX_CHARS, + ) + with gr.Column(scale=2): + say_voice = gr.Dropdown( + VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True + ) + say_lang = gr.Dropdown( + LANGUAGE_CHOICES, value="", label="Read the text as" + ) + with gr.Row(): + say_seed = gr.Number(value=7, precision=0, label="Seed") + say_speed = gr.Slider( + lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed" + ) + say_go = gr.Button("Speak", variant="primary") + say_out = gr.Audio(label="Speech", type="filepath") + say_stats = gr.Markdown() + + say_go.click( + speak, + [say, say_voice, say_lang, say_seed, say_speed], + [say_out, say_stats], + ) + + with gr.Accordion("Determinism check", open=False): + gr.Markdown( + "This renders the same text twice at the same seed and hashes " + "both waveforms. The digests must match." + ) + with gr.Row(): + probe_text = gr.Textbox( + value="The same seed gives the same audio.", + label="Text", + max_length=MAX_PROBE_CHARS, + scale=3, + ) + probe_voice = gr.Dropdown( + VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", scale=2 + ) + probe_seed = gr.Number(value=7, precision=0, label="Seed", scale=1) + probe_go = gr.Button("Render twice") + probe_out = gr.Markdown() + probe_go.click(probe, [probe_text, probe_voice, probe_seed], probe_out) + + # ---------------- Clone ---------------- + with gr.Tab("Clone"): + gr.Markdown( + f""" +Clone a voice from a short recording, then speak with it. + +- Record 5 to 10 seconds. Read anything. Speak normally. +- Clone only your own voice, or a voice you have permission to use. +- Nothing you record is kept. The recording is deleted when the request ends. +- See [Responsible use]({RESPONSIBLE_USE}). +""" + ) + with gr.Row(): + with gr.Column(scale=1): + mic = gr.Audio( + sources=["microphone"], + type="filepath", + label="Record yourself", + ) + with gr.Accordion("Upload a file instead", open=False): + upload = gr.Audio( + sources=["upload"], type="filepath", label="Audio file" + ) + consent = gr.Checkbox( + value=False, + label=( + "This is my own voice, or I have permission from the " + "person who owns it." + ), + ) + with gr.Column(scale=1): + clone_text = gr.Textbox( + label="Text to speak", + placeholder="Now in my own voice.", + lines=3, + max_length=MAX_CLONE_CHARS, + ) + clone_lang = gr.Dropdown( + LANGUAGE_CHOICES[1:], value="en", label="Language of the text" + ) + with gr.Row(): + clone_seed = gr.Number(value=7, precision=0, label="Seed") + clone_speed = gr.Slider( + lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed" + ) + clone_go = gr.Button("Clone and speak", variant="primary") + clone_out = gr.Audio(label="Speech", type="filepath") + clone_stats = gr.Markdown() + + clone_go.click( + clone, + [mic, upload, consent, clone_text, clone_lang, clone_seed, clone_speed], + [clone_out, clone_stats], + ) + + gr.Markdown( + f""" +--- +Run the same engine locally, where nothing is queued and nothing is metered. + +```bash +pip install "loudkit[torch,audio,enroll,hub]" +``` + +```python +import loudkit as lk + +engine = lk.load("{REPO}") +voice = lk.voice("kathleen", repo="{REPO}") +engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav") +``` + +Output files carry C2PA provenance: the fingerprint, the recipe and the seed. +""" + ) + +# The engine holds one set of weights and renders with an internal producer +# thread. One render at a time keeps two requests off the same buffers. +demo.queue(default_concurrency_limit=1, max_size=24) + +if __name__ == "__main__": + demo.launch() diff --git a/hf-loudkit/audio/carmen.opus b/hf-loudkit/audio/carmen.opus new file mode 100644 index 0000000000000000000000000000000000000000..e1e18609163ff3cd67315d85497818f104d33a58 Binary files /dev/null and b/hf-loudkit/audio/carmen.opus differ diff --git a/hf-loudkit/audio/colette.opus b/hf-loudkit/audio/colette.opus new file mode 100644 index 0000000000000000000000000000000000000000..682389d876f822ca755d906f05db376a18ff68d9 Binary files /dev/null and b/hf-loudkit/audio/colette.opus differ diff --git a/hf-loudkit/audio/dante.opus b/hf-loudkit/audio/dante.opus new file mode 100644 index 0000000000000000000000000000000000000000..6628f6858d4f71d379c47411baee9507f8192928 Binary files /dev/null and b/hf-loudkit/audio/dante.opus differ diff --git a/hf-loudkit/audio/darkman.opus b/hf-loudkit/audio/darkman.opus new file mode 100644 index 0000000000000000000000000000000000000000..7a6ce2e7c75e4206475b532df2fe8a9850a29929 Binary files /dev/null and b/hf-loudkit/audio/darkman.opus differ diff --git a/hf-loudkit/audio/dave.opus b/hf-loudkit/audio/dave.opus new file mode 100644 index 0000000000000000000000000000000000000000..502d2da2e6fdf140c6310a3336457eb210a7bc31 Binary files /dev/null and b/hf-loudkit/audio/dave.opus differ diff --git a/hf-loudkit/audio/freja.opus b/hf-loudkit/audio/freja.opus new file mode 100644 index 0000000000000000000000000000000000000000..2160a36de61672adf91284b9c74c692b59f07206 Binary files /dev/null and b/hf-loudkit/audio/freja.opus differ diff --git a/hf-loudkit/audio/gosia.opus b/hf-loudkit/audio/gosia.opus new file mode 100644 index 0000000000000000000000000000000000000000..853a342df5107a4894770834ef16a8539ecf1f6f Binary files /dev/null and b/hf-loudkit/audio/gosia.opus differ diff --git a/hf-loudkit/audio/henri.opus b/hf-loudkit/audio/henri.opus new file mode 100644 index 0000000000000000000000000000000000000000..97c0f1a5a9bd18e1516d97a70f26fe3a5b3ae0bb Binary files /dev/null and b/hf-loudkit/audio/henri.opus differ diff --git a/hf-loudkit/audio/ines.opus b/hf-loudkit/audio/ines.opus new file mode 100644 index 0000000000000000000000000000000000000000..2f988d7adb9dc0bd7041089ac470d0cbee02c832 Binary files /dev/null and b/hf-loudkit/audio/ines.opus differ diff --git a/hf-loudkit/audio/joe.opus b/hf-loudkit/audio/joe.opus new file mode 100644 index 0000000000000000000000000000000000000000..d35624d4b0cd92eb44d278737e92a65c95db9892 Binary files /dev/null and b/hf-loudkit/audio/joe.opus differ diff --git a/hf-loudkit/audio/kathleen.opus b/hf-loudkit/audio/kathleen.opus new file mode 100644 index 0000000000000000000000000000000000000000..afab998420c69de122c666484f135471301633ef Binary files /dev/null and b/hf-loudkit/audio/kathleen.opus differ diff --git a/hf-loudkit/audio/kerstin.opus b/hf-loudkit/audio/kerstin.opus new file mode 100644 index 0000000000000000000000000000000000000000..bc9b99ead23656b1e5f2ca7af21146bc1923b6ef Binary files /dev/null and b/hf-loudkit/audio/kerstin.opus differ diff --git a/hf-loudkit/audio/nathalie.opus b/hf-loudkit/audio/nathalie.opus new file mode 100644 index 0000000000000000000000000000000000000000..78d74b015d59796e31d798558ffe6c846d4e5619 Binary files /dev/null and b/hf-loudkit/audio/nathalie.opus differ diff --git a/hf-loudkit/audio/nils.opus b/hf-loudkit/audio/nils.opus new file mode 100644 index 0000000000000000000000000000000000000000..9f317ca8b1aba3b9aa3dd0e86719cda67125b6d8 Binary files /dev/null and b/hf-loudkit/audio/nils.opus differ diff --git a/hf-loudkit/audio/paola.opus b/hf-loudkit/audio/paola.opus new file mode 100644 index 0000000000000000000000000000000000000000..6f936ee395156750e75bbb580eb239a62b32eb3a Binary files /dev/null and b/hf-loudkit/audio/paola.opus differ diff --git a/hf-loudkit/audio/pim.opus b/hf-loudkit/audio/pim.opus new file mode 100644 index 0000000000000000000000000000000000000000..5d31d74acd05d2cc5a8b191d3c78f06387e1eb05 Binary files /dev/null and b/hf-loudkit/audio/pim.opus differ diff --git a/hf-loudkit/audio/refs/carmen.opus b/hf-loudkit/audio/refs/carmen.opus new file mode 100644 index 0000000000000000000000000000000000000000..220d1a403a3b535d284ce5c4008e3a5fcb736945 Binary files /dev/null and b/hf-loudkit/audio/refs/carmen.opus differ diff --git a/hf-loudkit/audio/refs/colette.opus b/hf-loudkit/audio/refs/colette.opus new file mode 100644 index 0000000000000000000000000000000000000000..dce18d4f09e5b07e10f9c46441fca118b4d17b67 Binary files /dev/null and b/hf-loudkit/audio/refs/colette.opus differ diff --git a/hf-loudkit/audio/refs/dante.opus b/hf-loudkit/audio/refs/dante.opus new file mode 100644 index 0000000000000000000000000000000000000000..b58368afe0db03b2d2f960c75d91280146d5f901 Binary files /dev/null and b/hf-loudkit/audio/refs/dante.opus differ diff --git a/hf-loudkit/audio/refs/darkman.opus b/hf-loudkit/audio/refs/darkman.opus new file mode 100644 index 0000000000000000000000000000000000000000..cb98a474936dbbdb79b3020552e955242f61ef0d Binary files /dev/null and b/hf-loudkit/audio/refs/darkman.opus differ diff --git a/hf-loudkit/audio/refs/dave.opus b/hf-loudkit/audio/refs/dave.opus new file mode 100644 index 0000000000000000000000000000000000000000..ca3c66e199e7bd900d1d4180aa91b65aaba5d9e8 Binary files /dev/null and b/hf-loudkit/audio/refs/dave.opus differ diff --git a/hf-loudkit/audio/refs/freja.opus b/hf-loudkit/audio/refs/freja.opus new file mode 100644 index 0000000000000000000000000000000000000000..adbb18927dc20821a83eb11190e9faa2aa50efb3 Binary files /dev/null and b/hf-loudkit/audio/refs/freja.opus differ diff --git a/hf-loudkit/audio/refs/gosia.opus b/hf-loudkit/audio/refs/gosia.opus new file mode 100644 index 0000000000000000000000000000000000000000..ac8e283877021b29edb5bd8e27c296973cae3c07 Binary files /dev/null and b/hf-loudkit/audio/refs/gosia.opus differ diff --git a/hf-loudkit/audio/refs/henri.opus b/hf-loudkit/audio/refs/henri.opus new file mode 100644 index 0000000000000000000000000000000000000000..59d6873d132f1a4b61081a27cb1817fa1d463bbe Binary files /dev/null and b/hf-loudkit/audio/refs/henri.opus differ diff --git a/hf-loudkit/audio/refs/ines.opus b/hf-loudkit/audio/refs/ines.opus new file mode 100644 index 0000000000000000000000000000000000000000..b31012129731dabfcdd0829dabb54a8e09d5914f Binary files /dev/null and b/hf-loudkit/audio/refs/ines.opus differ diff --git a/hf-loudkit/audio/refs/joe.opus b/hf-loudkit/audio/refs/joe.opus new file mode 100644 index 0000000000000000000000000000000000000000..732465cbc480985e84fe73887ca79a520a33e374 Binary files /dev/null and b/hf-loudkit/audio/refs/joe.opus differ diff --git a/hf-loudkit/audio/refs/kathleen.opus b/hf-loudkit/audio/refs/kathleen.opus new file mode 100644 index 0000000000000000000000000000000000000000..03fa0d9136a7614ef66b4d9627042992e6904be2 Binary files /dev/null and b/hf-loudkit/audio/refs/kathleen.opus differ diff --git a/hf-loudkit/audio/refs/kerstin.opus b/hf-loudkit/audio/refs/kerstin.opus new file mode 100644 index 0000000000000000000000000000000000000000..d6ca625df4b5c103120b206c35af9cc2c7b47936 Binary files /dev/null and b/hf-loudkit/audio/refs/kerstin.opus differ diff --git a/hf-loudkit/audio/refs/nathalie.opus b/hf-loudkit/audio/refs/nathalie.opus new file mode 100644 index 0000000000000000000000000000000000000000..5bc6b148d91e82fa1caa46ce33993fd078319262 Binary files /dev/null and b/hf-loudkit/audio/refs/nathalie.opus differ diff --git a/hf-loudkit/audio/refs/nils.opus b/hf-loudkit/audio/refs/nils.opus new file mode 100644 index 0000000000000000000000000000000000000000..bdfc6564eeeb9a75cf8f03f202369a2ae0604380 Binary files /dev/null and b/hf-loudkit/audio/refs/nils.opus differ diff --git a/hf-loudkit/audio/refs/paola.opus b/hf-loudkit/audio/refs/paola.opus new file mode 100644 index 0000000000000000000000000000000000000000..cef273a388c73c7e2e3f370ce4af34b56c954344 Binary files /dev/null and b/hf-loudkit/audio/refs/paola.opus differ diff --git a/hf-loudkit/audio/refs/pim.opus b/hf-loudkit/audio/refs/pim.opus new file mode 100644 index 0000000000000000000000000000000000000000..df2d2719e8e384f380a958142f3d700406ce6e61 Binary files /dev/null and b/hf-loudkit/audio/refs/pim.opus differ diff --git a/hf-loudkit/audio/refs/selma.opus b/hf-loudkit/audio/refs/selma.opus new file mode 100644 index 0000000000000000000000000000000000000000..d6bfaab9d0c0da5e7fe91eb4d2e799e388c21c89 Binary files /dev/null and b/hf-loudkit/audio/refs/selma.opus differ diff --git a/hf-loudkit/audio/refs/soren.opus b/hf-loudkit/audio/refs/soren.opus new file mode 100644 index 0000000000000000000000000000000000000000..dda8de236b964dbe5fd8b28e1311cb7fe65540d8 Binary files /dev/null and b/hf-loudkit/audio/refs/soren.opus differ diff --git a/hf-loudkit/audio/refs/thorsten.opus b/hf-loudkit/audio/refs/thorsten.opus new file mode 100644 index 0000000000000000000000000000000000000000..5e89b94fbd091c56363285eeebc1fe81deedccac Binary files /dev/null and b/hf-loudkit/audio/refs/thorsten.opus differ diff --git a/hf-loudkit/audio/refs/tugao.opus b/hf-loudkit/audio/refs/tugao.opus new file mode 100644 index 0000000000000000000000000000000000000000..3df5f5af70b92a74f4a006e1a75581f82d027328 Binary files /dev/null and b/hf-loudkit/audio/refs/tugao.opus differ diff --git a/hf-loudkit/audio/selma.opus b/hf-loudkit/audio/selma.opus new file mode 100644 index 0000000000000000000000000000000000000000..635b80f10f7832625903ac39d8bf0e831df43eb8 Binary files /dev/null and b/hf-loudkit/audio/selma.opus differ diff --git a/hf-loudkit/audio/soren.opus b/hf-loudkit/audio/soren.opus new file mode 100644 index 0000000000000000000000000000000000000000..2ed33a17736947e4ed72c1b4c5fc05b93cc548a9 Binary files /dev/null and b/hf-loudkit/audio/soren.opus differ diff --git a/hf-loudkit/audio/thorsten.opus b/hf-loudkit/audio/thorsten.opus new file mode 100644 index 0000000000000000000000000000000000000000..278cbffa5785f98db6cd5fa1394f1eff47e6f5ef Binary files /dev/null and b/hf-loudkit/audio/thorsten.opus differ diff --git a/hf-loudkit/audio/tugao.opus b/hf-loudkit/audio/tugao.opus new file mode 100644 index 0000000000000000000000000000000000000000..71f242f67742e53eecfaeb6363a4779696d3d48c Binary files /dev/null and b/hf-loudkit/audio/tugao.opus differ diff --git a/hf-loudkit/hf-loudkit/.gitattributes b/hf-loudkit/hf-loudkit/.gitattributes new file mode 100644 index 0000000000000000000000000000000000000000..a6344aac8c09253b3b630fb776ae94478aa0275b --- /dev/null +++ b/hf-loudkit/hf-loudkit/.gitattributes @@ -0,0 +1,35 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text diff --git a/hf-loudkit/hf-loudkit/README.md b/hf-loudkit/hf-loudkit/README.md new file mode 100644 index 0000000000000000000000000000000000000000..4db28afa9528f68415538245009b70427e84e4f0 --- /dev/null +++ b/hf-loudkit/hf-loudkit/README.md @@ -0,0 +1,77 @@ +--- +title: loudkit +emoji: 🔊 +colorFrom: gray +colorTo: red +sdk: gradio +sdk_version: 5.50.0 +python_version: "3.12.12" +app_file: app.py +pinned: false +license: apache-2.0 +short_description: On-device TTS. Twenty voices, ten languages, one engine. +models: + - loudreader/loudr-1 +preload_from_hub: + - loudreader/loudr-1 loudr-1.safetensors,loudr-1-enrollment.safetensors,ve.safetensors,manifest.json,release.json,SHA256SUMS,voices/carmen.safetensors,voices/colette.safetensors,voices/dante.safetensors,voices/darkman.safetensors,voices/dave.safetensors,voices/freja.safetensors,voices/gosia.safetensors,voices/henri.safetensors,voices/ines.safetensors,voices/joe.safetensors,voices/kathleen.safetensors,voices/kerstin.safetensors,voices/nathalie.safetensors,voices/nils.safetensors,voices/paola.safetensors,voices/pim.safetensors,voices/selma.safetensors,voices/soren.safetensors,voices/thorsten.safetensors,voices/tugao.safetensors +--- + +# loudkit + +Twenty voices across ten languages, from [loudreader/loudr-1](https://huggingface.co/loudreader/loudr-1), +running the [loudkit](https://github.com/loudreader/loudkit) engine on ZeroGPU. + +## Three tabs + +- **Listen.** Twenty voices, each beside the reference recording it was enrolled + from. These files were rendered ahead of time and ship in this repo. This tab + uses no GPU. +- **Speak.** Your text, in any of the twenty voices. Up to 1,000 characters. +- **Clone.** Your own voice, from about ten seconds of audio. + +ZeroGPU bills GPU time to the visitor, not to the owner. An anonymous visitor +gets about two minutes a day. A signed-in free account gets about five. Listening +costs none of it. + +## Cloning and consent + +Clone your own voice, or a voice you have permission to use. + +- The microphone is the default path. +- An upload is secondary, and needs an explicit confirmation. +- Recordings are deleted when the request ends. Nothing is kept. + +See [RESPONSIBLE_USE](https://github.com/loudreader/loudkit/blob/main/RESPONSIBLE_USE.md). + +## Determinism + +The Speak tab has a determinism check. It renders the same text twice at the same +seed and prints the SHA-256 of both waveforms. They match. + +That holds within this build and this device. loudkit promises a bit-identical +waveform for the same seed, build, backend and input. It does not promise that +your machine matches this GPU. See the +[identity contract](https://github.com/loudreader/loudkit/blob/main/docs/reference/IDENTITY-CONTRACT.md). + +## Run it locally + +```bash +pip install "loudkit[torch,audio,enroll,hub]" +``` + +```python +import loudkit as lk + +engine = lk.load("loudreader/loudr-1") +voice = lk.voice("kathleen", repo="loudreader/loudr-1") +engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav") +``` + +Audio in this Space, and from `Result.save`, carries C2PA provenance: the +algorithm fingerprint, the recipe and the seed. + +## Voice sources + +Every voice is enrolled from a public-domain or openly licensed recording. +`voices.json` in this repo carries the full record for each one: donor, source, +licence, consent, and the SHA-256 of both the reference and the sample. diff --git a/hf-loudkit/hf-loudkit/app.py b/hf-loudkit/hf-loudkit/app.py new file mode 100644 index 0000000000000000000000000000000000000000..3e2e44e07163d6b6d66f366ce80678012a99f0f9 --- /dev/null +++ b/hf-loudkit/hf-loudkit/app.py @@ -0,0 +1,541 @@ +"""The loudkit demo Space: hear twenty voices, speak your own text, clone your own. + +ZeroGPU bills GPU time to the *visitor*, not to the owner: an anonymous visitor +gets about two minutes a day, a signed-in free account about five. A demo whose +first click spends that budget is one most people bounce off before they have +heard anything at all. So the Listen tab is twenty pre-rendered files served +straight out of this repo — no GPU, no queue, no quota — and the GPU is spent +only on what a visitor types or records. + +The engine and the enroller are both built on `cuda` at module level, which is +what ZeroGPU asks for: CUDA transfers are optimised for start-up placement, and +lazy-loading inside a `@spaces.GPU` function is explicitly discouraged. Each +decorated call then runs in a freshly forked, short-lived process, which is also +why there is no `torch.compile` and no CUDA graph capture here: both pay their +cost once per process and would never amortise. + +Cloning is exposed, which the CPU scaffold this replaces deliberately did not do. +The reasoning that kept it out was about consent, not about capability, so the +consent is built into the shape of the tab rather than written beside it: the +microphone is the default path, an upload is secondary and gated on an explicit +confirmation, and neither recording outlives the request that carried it. +""" + +from __future__ import annotations + +import contextlib +import dataclasses +import hashlib +import json +import os +import tempfile +from pathlib import Path + +# Before torch, and before anything that imports torch. The module installs the +# CUDA emulation that lets a module-level `.to("cuda")` succeed on a machine +# that has no GPU attached yet. +import spaces + +import gradio as gr +import numpy as np + +import loudkit as lk +from loudkit.backends.torch_backend import build_torch_enroller +from loudkit.hub import resolve_enrollment_checkpoint, resolve_voice_encoder + +REPO = "loudreader/loudr-1" +DEVICE = "cuda" +HERE = Path(__file__).parent + +# The CPU scaffold capped text at 300 characters because CPU synthesis ran at +# roughly a tenth of real time. On a GPU the cap is about the visitor's daily +# quota instead, which is a far looser bound: 1000 characters is ~70 s of speech. +MAX_CHARS = 1_000 +MAX_CLONE_CHARS = 400 +MAX_PROBE_CHARS = 200 + +# The enroller refuses anything over 30 s and wants 5 to 10. The prompt is built +# from the first 10 s; the speaker embedding reads whatever else is there, so a +# little past the prompt window is useful and 20 s stays clear of the refusal. +ENROLL_SECONDS = 20.0 + +DOCS = "https://github.com/loudreader/loudkit" +IDENTITY_CONTRACT = f"{DOCS}/blob/main/docs/reference/IDENTITY-CONTRACT.md" +RESPONSIBLE_USE = f"{DOCS}/blob/main/RESPONSIBLE_USE.md" + +ROSTER = json.loads((HERE / "voices.json").read_text(encoding="utf-8")) +BY_NAME = {entry["name"]: entry for entry in ROSTER} +ORDERED = sorted(ROSTER, key=lambda e: (e["language"], e["name"])) +VOICE_CHOICES = [(f"{e['name']} · {e['language']} ({e['gender']})", e["name"]) for e in ORDERED] + +# -------------------------------------------------------------------------- +# Module-level model placement, per the ZeroGPU contract. +# -------------------------------------------------------------------------- + +engine = lk.load(REPO, device=DEVICE) + +# Voice profiles are numpy, not torch, so they are device-agnostic and cost a +# few hundred kilobytes each. Loading all twenty up front means switching voice +# in the Speak tab never blocks on a download. +PROFILES = {entry["name"]: lk.voice(entry["name"], repo=REPO) for entry in ROSTER} + +# Enrollment reads the other half of the release: the speech tokenizer and the +# speaker encoder, which synthesis never touches, plus the utterance voice +# encoder that sits beside both. `lk.enroll()` builds this per call by design; +# a Space would pay the load on every clone, so it is built once here instead. +enroller = build_torch_enroller( + str(resolve_enrollment_checkpoint(REPO)), + device=DEVICE, + voice_encoder_weights=str(resolve_voice_encoder(REPO)), +) + +FINGERPRINT = engine.algorithm.fingerprint() + +_LANGUAGE_NAMES = {e["language_id"]: e["language"] for e in ROSTER} +LANGUAGE_CHOICES = [("Follow the voice", "")] + [ + (f"{_LANGUAGE_NAMES.get(code, code)} ({code})", code) for code in lk.languages() +] + +# -------------------------------------------------------------------------- +# Helpers +# -------------------------------------------------------------------------- + + +def _sha256_audio(audio: np.ndarray) -> str: + """Hash the waveform, not the file. + + `Result.save` appends a C2PA manifest carrying a wall-clock creation time, + which the library itself calls the one byte range in which two identical + renders may legitimately differ. Hashing the saved WAV would therefore print + two different digests for two identical renders and read as a determinism + failure. The waveform is what the identity contract makes its promise about, + so the waveform is what gets hashed. + """ + return hashlib.sha256(np.ascontiguousarray(audio, dtype=np.float32).tobytes()).hexdigest() + + +def _write(result: lk.Result, *, voice: str, language: str) -> str: + out = tempfile.NamedTemporaryFile(suffix=".wav", delete=False) + out.close() + # Provenance on: the manifest carries the fingerprint, the recipe and the + # seed, which is the machine-readable marking a synthetic-speech demo should + # be handing out by default. + result.save(out.name, voice=voice, language=language) + return out.name + + +def _stats(result: lk.Result) -> str: + seconds = len(result.audio) / result.sample_rate + return ( + f"**{seconds:.1f} s of audio.** {result.timings.describe(seconds)}\n\n" + f"`algo[{result.algorithm_fingerprint}]` · seed `{result.seed}` · " + f"speed `{result.speed:g}x` · {result.sample_rate} Hz" + ) + + +def _estimate(text: str, *, passes: int = 1, overhead: float = 15.0) -> int: + """Seconds of GPU to ask for. + + Speech runs at roughly 14 characters a second, and the render is asked to + keep up with better than real time; the overhead covers the process fork and + the first real CUDA touch. Asking for too much costs queue priority but not + quota, which is charged on effective duration, so this leans generous. + """ + audio_seconds = len((text or "").strip()) / 14.0 + return int(min(180.0, overhead + passes * max(4.0, audio_seconds * 0.9))) + + +def _check(text: str, limit: int) -> str: + text = (text or "").strip() + if not text: + raise gr.Error("Type something to say.") + if len(text) > limit: + raise gr.Error(f"Keep it under {limit:,} characters here. The library itself takes 10,000.") + return text + + +# -------------------------------------------------------------------------- +# Listen. No GPU: these files were rendered ahead of time and ship in the repo. +# -------------------------------------------------------------------------- + + +def listen(name: str): + entry = BY_NAME[name] + sample, reference, source = entry["sample"], entry["reference"], entry["source"] + + lines = [ + f"### {entry['name']}. {entry['language']} ({entry['gender']}).", + "", + f"> {sample['text']}", + "", + f"From *{sample['work']}*, seed `{sample['seed']}`.", + "", + f"- Reference recording: {reference['duration_s']:.1f} s, {reference['construction']}.", + f"- Source: [{source['name']}]({source['url']}), {source['license']}.", + f"- Consent: {source['consent']}.", + ] + similarity = entry.get("speaker_similarity") + if similarity is not None: + lines.append(f"- Speaker similarity to the reference: {similarity:.3f}.") + lines.append(f"- Voice profile: `{entry['profile']['hf_path']}`.") + + return ( + str(HERE / sample["audio"]), + str(HERE / reference["public_preview"]), + "\n".join(lines), + ) + + +ROSTER_TABLE = [ + [ + entry["name"], + entry["language"], + entry["gender"], + entry["source"]["license"], + f"{entry['speaker_similarity']:.3f}" if entry.get("speaker_similarity") is not None else "", + ] + for entry in ORDERED +] + + +# -------------------------------------------------------------------------- +# Speak. GPU. +# -------------------------------------------------------------------------- + + +def _speak_duration(text, name, language, seed, speed): + return _estimate(text, overhead=15.0) + + +@spaces.GPU(duration=_speak_duration) +def speak(text: str, name: str, language: str, seed: float, speed: float): + text = _check(text, MAX_CHARS) + result = engine.synthesize_long( + text, + PROFILES[name], + seed=int(seed), + language=language or None, + speed=float(speed), + ) + label = language or BY_NAME[name]["language_id"] + return _write(result, voice=name, language=label), _stats(result) + + +# -------------------------------------------------------------------------- +# Clone. GPU. The microphone is the default path; an upload is gated. +# -------------------------------------------------------------------------- + + +def _clone_duration(mic, upload, consent, text, language, seed, speed): + # Enrollment is a fixed cost on top of the render: two encoders and a + # tokenizer over at most 20 s of audio. + return _estimate(text, overhead=30.0) + + +@spaces.GPU(duration=_clone_duration) +def clone(mic, upload, consent: bool, text: str, language: str, seed: float, speed: float): + source = mic or upload + if not source: + raise gr.Error("Record yourself first, or upload a clip you are allowed to use.") + if upload and not mic and not consent: + raise gr.Error("Confirm the uploaded voice is yours, or that you have permission to use it.") + text = _check(text, MAX_CLONE_CHARS) + + try: + import librosa + + samples, _ = librosa.load(source, sr=24_000, mono=True) + limit = int(ENROLL_SECONDS * 24_000) + if samples.size > limit: + samples = samples[:limit] + + try: + profile = enroller.enroll(samples, 24_000, name="your voice") + except ValueError as exc: + # The library's own messages name the bound and describe a good + # input, which is more useful than anything restated here. + raise gr.Error(str(exc)) from exc + + # `enroll` writes no language, so every cloned voice would claim English + # and read its text through the English funnel. + profile = dataclasses.replace(profile, language=language or "en") + + result = engine.synthesize_long( + text, profile, seed=int(seed), language=language or None, speed=float(speed) + ) + return _write(result, voice="cloned", language=profile.language), _stats(result) + finally: + # Nothing the visitor recorded outlives the request that carried it. + with contextlib.suppress(OSError): + os.unlink(source) + + +# -------------------------------------------------------------------------- +# Determinism probe. GPU. Renders the same text twice at the same seed. +# -------------------------------------------------------------------------- + + +def _probe_duration(text, name, seed): + return _estimate(text, passes=2, overhead=20.0) + + +@spaces.GPU(duration=_probe_duration) +def probe(text: str, name: str, seed: float): + text = _check(text, MAX_PROBE_CHARS) + profile = PROFILES[name] + first = engine.synthesize_long(text, profile, seed=int(seed)) + second = engine.synthesize_long(text, profile, seed=int(seed)) + + left, right = _sha256_audio(first.audio), _sha256_audio(second.audio) + verdict = "Identical." if left == right else "Different. Please report this." + + return "\n".join( + [ + f"**{verdict}**", + "", + "```", + f"render 1 sha256 {left}", + f"render 2 sha256 {right}", + f" algo[{first.algorithm_fingerprint}] seed {int(seed)}", + "```", + "", + "Identical within this build and this device. loudkit promises a " + "bit-identical waveform for the same seed, build, backend and input. " + "It does not promise that your laptop matches this GPU. " + f"[Read the identity contract]({IDENTITY_CONTRACT}).", + ] + ) + + +# -------------------------------------------------------------------------- +# Interface +# -------------------------------------------------------------------------- + +# loudreader.io: cream ground, ink text, black pill buttons at 14px. +CSS = """ +#lk-head h1 { font-size: 2.15rem; margin-bottom: .25rem; letter-spacing: -.02em; } +#lk-head p { margin-top: 0; } +.lk-pill { + display: inline-block; padding: .2rem .75rem; margin: .15rem .35rem .15rem 0; + border: 1px solid #ded8ce; border-radius: 999px; font-size: .8rem; + color: #374151; background: #fffdfa; +} +.lk-card { background: #fffdfa; border: 1px solid #e7e1d7; border-radius: 14px; padding: .35rem 1rem; } +footer { display: none !important; } +""" + +# Gradio follows the visitor's system theme unless told otherwise, and this +# palette is light-first. Without this the ink-on-cream tokens below land under +# a dark stylesheet and the text turns near-white on a cream ground. +FORCE_LIGHT = """ +() => { + const url = new URL(window.location); + if (url.searchParams.get('__theme') !== 'light') { + url.searchParams.set('__theme', 'light'); + window.location.replace(url.href); + } +} +""" + +THEME = gr.themes.Soft( + primary_hue=gr.themes.colors.gray, + neutral_hue=gr.themes.colors.stone, + font=[gr.themes.GoogleFont("Inter"), "system-ui", "sans-serif"], +).set( + body_background_fill="#f7f5f2", + body_text_color="#111827", + body_text_color_subdued="#4b5563", + block_background_fill="#fffdfa", + block_border_color="#e7e1d7", + border_color_primary="#e7e1d7", + input_background_fill="#ffffff", + button_primary_background_fill="#111827", + button_primary_background_fill_hover="#374151", + button_primary_text_color="#ffffff", + button_large_radius="14px", + button_small_radius="14px", +) + +with gr.Blocks(title="loudkit", theme=THEME, css=CSS, js=FORCE_LIGHT, fill_width=False) as demo: + gr.Markdown( + f""" +# Twenty voices. Ten languages. One engine. + +On-device text to speech, running here on ZeroGPU. +[Model](https://huggingface.co/{REPO}) · [Code]({DOCS}) · [Responsible use]({RESPONSIBLE_USE}) + +Listening costs no GPU +Speaking and cloning spend your daily quota +algo[{FINGERPRINT}] +""", + elem_id="lk-head", + ) + + with gr.Tabs(): + # ---------------- Listen ---------------- + with gr.Tab("Listen"): + gr.Markdown( + "Twenty voices, rendered ahead of time and served as files. " + "This tab uses no GPU and spends none of your quota. " + "Each voice is paired with the reference recording it was enrolled from." + ) + with gr.Row(): + with gr.Column(scale=1): + pick = gr.Dropdown( + VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True + ) + made = gr.Audio(label="loudkit", type="filepath", interactive=False) + ref = gr.Audio(label="Reference recording", type="filepath", interactive=False) + with gr.Column(scale=1): + card = gr.Markdown(elem_classes="lk-card") + + with gr.Accordion("The whole roster", open=False): + gr.Dataframe( + value=ROSTER_TABLE, + headers=["Voice", "Language", "Gender", "Licence", "Similarity"], + interactive=False, + wrap=True, + ) + + pick.change(listen, pick, [made, ref, card]) + demo.load(listen, pick, [made, ref, card]) + + # ---------------- Speak ---------------- + with gr.Tab("Speak"): + gr.Markdown( + f"Your text, in one of the twenty voices. " + f"Up to {MAX_CHARS:,} characters here. The library itself takes 10,000. " + "This tab spends your ZeroGPU quota." + ) + with gr.Row(): + with gr.Column(scale=3): + say = gr.Textbox( + label="Text", + placeholder="Hello from loudkit.", + lines=4, + max_length=MAX_CHARS, + ) + with gr.Column(scale=2): + say_voice = gr.Dropdown( + VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True + ) + say_lang = gr.Dropdown( + LANGUAGE_CHOICES, value="", label="Read the text as" + ) + with gr.Row(): + say_seed = gr.Number(value=7, precision=0, label="Seed") + say_speed = gr.Slider( + lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed" + ) + say_go = gr.Button("Speak", variant="primary") + say_out = gr.Audio(label="Speech", type="filepath") + say_stats = gr.Markdown() + + say_go.click( + speak, + [say, say_voice, say_lang, say_seed, say_speed], + [say_out, say_stats], + ) + + with gr.Accordion("Determinism check", open=False): + gr.Markdown( + "This renders the same text twice at the same seed and hashes " + "both waveforms. The digests must match." + ) + with gr.Row(): + probe_text = gr.Textbox( + value="The same seed gives the same audio.", + label="Text", + max_length=MAX_PROBE_CHARS, + scale=3, + ) + probe_voice = gr.Dropdown( + VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", scale=2 + ) + probe_seed = gr.Number(value=7, precision=0, label="Seed", scale=1) + probe_go = gr.Button("Render twice") + probe_out = gr.Markdown() + probe_go.click(probe, [probe_text, probe_voice, probe_seed], probe_out) + + # ---------------- Clone ---------------- + with gr.Tab("Clone"): + gr.Markdown( + f""" +Clone a voice from a short recording, then speak with it. + +- Record 5 to 10 seconds. Read anything. Speak normally. +- Clone only your own voice, or a voice you have permission to use. +- Nothing you record is kept. The recording is deleted when the request ends. +- See [Responsible use]({RESPONSIBLE_USE}). +""" + ) + with gr.Row(): + with gr.Column(scale=1): + mic = gr.Audio( + sources=["microphone"], + type="filepath", + label="Record yourself", + ) + with gr.Accordion("Upload a file instead", open=False): + upload = gr.Audio( + sources=["upload"], type="filepath", label="Audio file" + ) + consent = gr.Checkbox( + value=False, + label=( + "This is my own voice, or I have permission from the " + "person who owns it." + ), + ) + with gr.Column(scale=1): + clone_text = gr.Textbox( + label="Text to speak", + placeholder="Now in my own voice.", + lines=3, + max_length=MAX_CLONE_CHARS, + ) + clone_lang = gr.Dropdown( + LANGUAGE_CHOICES[1:], value="en", label="Language of the text" + ) + with gr.Row(): + clone_seed = gr.Number(value=7, precision=0, label="Seed") + clone_speed = gr.Slider( + lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed" + ) + clone_go = gr.Button("Clone and speak", variant="primary") + clone_out = gr.Audio(label="Speech", type="filepath") + clone_stats = gr.Markdown() + + clone_go.click( + clone, + [mic, upload, consent, clone_text, clone_lang, clone_seed, clone_speed], + [clone_out, clone_stats], + ) + + gr.Markdown( + f""" +--- +Run the same engine locally, where nothing is queued and nothing is metered. + +```bash +pip install "loudkit[torch,audio,enroll,hub]" +``` + +```python +import loudkit as lk + +engine = lk.load("{REPO}") +voice = lk.voice("kathleen", repo="{REPO}") +engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav") +``` + +Output files carry C2PA provenance: the fingerprint, the recipe and the seed. +""" + ) + +# The engine holds one set of weights and renders with an internal producer +# thread. One render at a time keeps two requests off the same buffers. +demo.queue(default_concurrency_limit=1, max_size=24) + +if __name__ == "__main__": + demo.launch() diff --git a/hf-loudkit/hf-loudkit/audio/carmen.opus b/hf-loudkit/hf-loudkit/audio/carmen.opus new file mode 100644 index 0000000000000000000000000000000000000000..e1e18609163ff3cd67315d85497818f104d33a58 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/carmen.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/colette.opus b/hf-loudkit/hf-loudkit/audio/colette.opus new file mode 100644 index 0000000000000000000000000000000000000000..682389d876f822ca755d906f05db376a18ff68d9 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/colette.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/dante.opus b/hf-loudkit/hf-loudkit/audio/dante.opus new file mode 100644 index 0000000000000000000000000000000000000000..6628f6858d4f71d379c47411baee9507f8192928 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/dante.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/darkman.opus b/hf-loudkit/hf-loudkit/audio/darkman.opus new file mode 100644 index 0000000000000000000000000000000000000000..7a6ce2e7c75e4206475b532df2fe8a9850a29929 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/darkman.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/dave.opus b/hf-loudkit/hf-loudkit/audio/dave.opus new file mode 100644 index 0000000000000000000000000000000000000000..502d2da2e6fdf140c6310a3336457eb210a7bc31 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/dave.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/freja.opus b/hf-loudkit/hf-loudkit/audio/freja.opus new file mode 100644 index 0000000000000000000000000000000000000000..2160a36de61672adf91284b9c74c692b59f07206 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/freja.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/gosia.opus b/hf-loudkit/hf-loudkit/audio/gosia.opus new file mode 100644 index 0000000000000000000000000000000000000000..853a342df5107a4894770834ef16a8539ecf1f6f Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/gosia.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/henri.opus b/hf-loudkit/hf-loudkit/audio/henri.opus new file mode 100644 index 0000000000000000000000000000000000000000..97c0f1a5a9bd18e1516d97a70f26fe3a5b3ae0bb Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/henri.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/ines.opus b/hf-loudkit/hf-loudkit/audio/ines.opus new file mode 100644 index 0000000000000000000000000000000000000000..2f988d7adb9dc0bd7041089ac470d0cbee02c832 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/ines.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/joe.opus b/hf-loudkit/hf-loudkit/audio/joe.opus new file mode 100644 index 0000000000000000000000000000000000000000..d35624d4b0cd92eb44d278737e92a65c95db9892 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/joe.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/kathleen.opus b/hf-loudkit/hf-loudkit/audio/kathleen.opus new file mode 100644 index 0000000000000000000000000000000000000000..afab998420c69de122c666484f135471301633ef Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/kathleen.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/kerstin.opus b/hf-loudkit/hf-loudkit/audio/kerstin.opus new file mode 100644 index 0000000000000000000000000000000000000000..bc9b99ead23656b1e5f2ca7af21146bc1923b6ef Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/kerstin.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/nathalie.opus b/hf-loudkit/hf-loudkit/audio/nathalie.opus new file mode 100644 index 0000000000000000000000000000000000000000..78d74b015d59796e31d798558ffe6c846d4e5619 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/nathalie.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/nils.opus b/hf-loudkit/hf-loudkit/audio/nils.opus new file mode 100644 index 0000000000000000000000000000000000000000..9f317ca8b1aba3b9aa3dd0e86719cda67125b6d8 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/nils.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/paola.opus b/hf-loudkit/hf-loudkit/audio/paola.opus new file mode 100644 index 0000000000000000000000000000000000000000..6f936ee395156750e75bbb580eb239a62b32eb3a Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/paola.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/pim.opus b/hf-loudkit/hf-loudkit/audio/pim.opus new file mode 100644 index 0000000000000000000000000000000000000000..5d31d74acd05d2cc5a8b191d3c78f06387e1eb05 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/pim.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/carmen.opus b/hf-loudkit/hf-loudkit/audio/refs/carmen.opus new file mode 100644 index 0000000000000000000000000000000000000000..220d1a403a3b535d284ce5c4008e3a5fcb736945 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/carmen.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/colette.opus b/hf-loudkit/hf-loudkit/audio/refs/colette.opus new file mode 100644 index 0000000000000000000000000000000000000000..dce18d4f09e5b07e10f9c46441fca118b4d17b67 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/colette.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/dante.opus b/hf-loudkit/hf-loudkit/audio/refs/dante.opus new file mode 100644 index 0000000000000000000000000000000000000000..b58368afe0db03b2d2f960c75d91280146d5f901 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/dante.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/darkman.opus b/hf-loudkit/hf-loudkit/audio/refs/darkman.opus new file mode 100644 index 0000000000000000000000000000000000000000..cb98a474936dbbdb79b3020552e955242f61ef0d Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/darkman.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/dave.opus b/hf-loudkit/hf-loudkit/audio/refs/dave.opus new file mode 100644 index 0000000000000000000000000000000000000000..ca3c66e199e7bd900d1d4180aa91b65aaba5d9e8 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/dave.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/freja.opus b/hf-loudkit/hf-loudkit/audio/refs/freja.opus new file mode 100644 index 0000000000000000000000000000000000000000..adbb18927dc20821a83eb11190e9faa2aa50efb3 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/freja.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/gosia.opus b/hf-loudkit/hf-loudkit/audio/refs/gosia.opus new file mode 100644 index 0000000000000000000000000000000000000000..ac8e283877021b29edb5bd8e27c296973cae3c07 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/gosia.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/henri.opus b/hf-loudkit/hf-loudkit/audio/refs/henri.opus new file mode 100644 index 0000000000000000000000000000000000000000..59d6873d132f1a4b61081a27cb1817fa1d463bbe Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/henri.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/ines.opus b/hf-loudkit/hf-loudkit/audio/refs/ines.opus new file mode 100644 index 0000000000000000000000000000000000000000..b31012129731dabfcdd0829dabb54a8e09d5914f Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/ines.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/joe.opus b/hf-loudkit/hf-loudkit/audio/refs/joe.opus new file mode 100644 index 0000000000000000000000000000000000000000..732465cbc480985e84fe73887ca79a520a33e374 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/joe.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/kathleen.opus b/hf-loudkit/hf-loudkit/audio/refs/kathleen.opus new file mode 100644 index 0000000000000000000000000000000000000000..03fa0d9136a7614ef66b4d9627042992e6904be2 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/kathleen.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/kerstin.opus b/hf-loudkit/hf-loudkit/audio/refs/kerstin.opus new file mode 100644 index 0000000000000000000000000000000000000000..d6ca625df4b5c103120b206c35af9cc2c7b47936 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/kerstin.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/nathalie.opus b/hf-loudkit/hf-loudkit/audio/refs/nathalie.opus new file mode 100644 index 0000000000000000000000000000000000000000..5bc6b148d91e82fa1caa46ce33993fd078319262 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/nathalie.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/nils.opus b/hf-loudkit/hf-loudkit/audio/refs/nils.opus new file mode 100644 index 0000000000000000000000000000000000000000..bdfc6564eeeb9a75cf8f03f202369a2ae0604380 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/nils.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/paola.opus b/hf-loudkit/hf-loudkit/audio/refs/paola.opus new file mode 100644 index 0000000000000000000000000000000000000000..cef273a388c73c7e2e3f370ce4af34b56c954344 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/paola.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/pim.opus b/hf-loudkit/hf-loudkit/audio/refs/pim.opus new file mode 100644 index 0000000000000000000000000000000000000000..df2d2719e8e384f380a958142f3d700406ce6e61 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/pim.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/selma.opus b/hf-loudkit/hf-loudkit/audio/refs/selma.opus new file mode 100644 index 0000000000000000000000000000000000000000..d6bfaab9d0c0da5e7fe91eb4d2e799e388c21c89 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/selma.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/soren.opus b/hf-loudkit/hf-loudkit/audio/refs/soren.opus new file mode 100644 index 0000000000000000000000000000000000000000..dda8de236b964dbe5fd8b28e1311cb7fe65540d8 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/soren.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/thorsten.opus b/hf-loudkit/hf-loudkit/audio/refs/thorsten.opus new file mode 100644 index 0000000000000000000000000000000000000000..5e89b94fbd091c56363285eeebc1fe81deedccac Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/thorsten.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/refs/tugao.opus b/hf-loudkit/hf-loudkit/audio/refs/tugao.opus new file mode 100644 index 0000000000000000000000000000000000000000..3df5f5af70b92a74f4a006e1a75581f82d027328 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/tugao.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/selma.opus b/hf-loudkit/hf-loudkit/audio/selma.opus new file mode 100644 index 0000000000000000000000000000000000000000..635b80f10f7832625903ac39d8bf0e831df43eb8 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/selma.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/soren.opus b/hf-loudkit/hf-loudkit/audio/soren.opus new file mode 100644 index 0000000000000000000000000000000000000000..2ed33a17736947e4ed72c1b4c5fc05b93cc548a9 Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/soren.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/thorsten.opus b/hf-loudkit/hf-loudkit/audio/thorsten.opus new file mode 100644 index 0000000000000000000000000000000000000000..278cbffa5785f98db6cd5fa1394f1eff47e6f5ef Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/thorsten.opus differ diff --git a/hf-loudkit/hf-loudkit/audio/tugao.opus b/hf-loudkit/hf-loudkit/audio/tugao.opus new file mode 100644 index 0000000000000000000000000000000000000000..71f242f67742e53eecfaeb6363a4779696d3d48c Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/tugao.opus differ diff --git a/hf-loudkit/hf-loudkit/requirements.txt b/hf-loudkit/hf-loudkit/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..60b4bdb741c3b04aa1bb164e82cbfb4bbb071a24 --- /dev/null +++ b/hf-loudkit/hf-loudkit/requirements.txt @@ -0,0 +1,18 @@ +# The library under demo. 0.1.0 is on PyPI, so this installs from the registry +# rather than carrying a wheel beside the app. +# +# torch the cuda backend this Space runs on +# audio soundfile, for Result.save +# enroll torchaudio + librosa, for the Clone tab +# hub huggingface_hub, to resolve loudreader/loudr-1 +loudkit[torch,audio,enroll,hub]==0.1.0 + +# loudkit floors torch at >=2.4 and sets no ceiling, and ZeroGPU runs a fixed +# set of builds: 2.8.0, 2.9.1, 2.10.0, 2.11.0. Pinned here so the resolver +# cannot land outside that window. torchaudio tracks torch version for version. +torch==2.8.0 +torchaudio==2.8.0 + +# The ZeroGPU decorator. Present on ZeroGPU hardware already; named here so a +# local run or a duplicated Space installs it too. +spaces>=0.42 diff --git a/hf-loudkit/hf-loudkit/voices.json b/hf-loudkit/hf-loudkit/voices.json new file mode 100644 index 0000000000000000000000000000000000000000..ac89f1f9159d32cd3c35686f32448d57a27284b2 --- /dev/null +++ b/hf-loudkit/hf-loudkit/voices.json @@ -0,0 +1,683 @@ +[ + { + "name": "darkman", + "language": "Polish", + "language_id": "pl", + "gender": "M", + "donor": "darkman (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "darkman.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/darkman.opus", + "sha256": "d228db9d17b1ddb4ba450f16a6b8650bb0a8550521b78a914cb9e573cd24d576", + "duration_s": 9.78, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/darkman.safetensors", + "sha256": "78988a6769f90c209240f9f285eb9cdd760d1ba471cfa7d6e67db632599cfa1b" + }, + "speaker_similarity": 0.935, + "sample": { + "seed": 7, + "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.", + "work": "Quo Vadis (Henryk Sienkiewicz) — public domain", + "audio": "audio/darkman.opus", + "sha256": "b30f98cf0452e9c694fcbafc27b413041c87022238f5424fc1aa452c8eb156a5" + } + }, + { + "name": "gosia", + "language": "Polish", + "language_id": "pl", + "gender": "F", + "donor": "gosia (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "gosia.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/gosia.opus", + "sha256": "02c1d09db8b097c66dd8705747a6ef138596f5f91c3213b1cd430ae75ca2279a", + "duration_s": 10.86, + "construction": "concatenated long donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/gosia.safetensors", + "sha256": "e115cf64bfa0ad15770a50c0be8097312a58d41f84c7bb4f01ecbdbac5d21823" + }, + "speaker_similarity": 0.917, + "sample": { + "seed": 7, + "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.", + "work": "Quo Vadis (Henryk Sienkiewicz) — public domain", + "audio": "audio/gosia.opus", + "sha256": "4ba1107d7f807ac2e0941167d1bc4a7cfb2039a4b0482a3bc798db5eeefe17e5" + } + }, + { + "name": "joe", + "language": "English", + "language_id": "en", + "gender": "M", + "donor": "joe (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "joe.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/joe.opus", + "sha256": "591c8d6a5799a4266d029762dc0f671ac9f0455382aeef1e763ca206d6fdd164", + "duration_s": 10.99, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/joe.safetensors", + "sha256": "cfc3687fdb1f6b56fe13e252ca4a5bd681cef0702dc5cd56d4ebbc60e1aea6e8" + }, + "speaker_similarity": 0.924, + "sample": { + "seed": 11, + "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.", + "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain", + "audio": "audio/joe.opus", + "sha256": "83e0bf0a1f6b03e047de66871536c3ff77291041d0a3f2907b6990e580519fea" + } + }, + { + "name": "kathleen", + "language": "English", + "language_id": "en", + "gender": "F", + "donor": "kathleen (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "kathleen.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/kathleen.opus", + "sha256": "f43a24359a333e1f723c5f10d7dc799ec5880abf2b5b9321a7d5f8c7064d6f73", + "duration_s": 11.0, + "construction": "concatenated donation clips; prompt texts from the CMU ARCTIC prompt list (public-domain Gutenberg novels), recording is kathleen's own CC0 donation" + }, + "profile": { + "hf_path": "voices/kathleen.safetensors", + "sha256": "6a845e4eb11989fc9bd8e19eed7c5ac585e6d08e1ec824211f0eb240ce523d29" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.", + "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain", + "audio": "audio/kathleen.opus", + "sha256": "48615517c1d1d2d55dbb7f74f37982918236e33f079947459d56c116cd70b7a3" + } + }, + { + "name": "thorsten", + "language": "German", + "language_id": "de", + "gender": "M", + "donor": "Thorsten Mueller", + "speaker_id": null, + "source": { + "name": "Thorsten-Voice TV-44kHz-Full", + "url": "https://huggingface.co/datasets/Thorsten-Voice/TV-44kHz-Full", + "license": "CC0-1.0", + "consent": "voice deliberately donated by Thorsten Mueller for TTS" + }, + "reference": { + "source_filename": "thorsten.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/thorsten.opus", + "sha256": "37805d1136f727f7530e714b772c23224ae1c6f6931ea5b78399cf286694de3e", + "duration_s": 6.37, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/thorsten.safetensors", + "sha256": "87873fec8d3dceaa643ab1ab07cf5143364e659dd6b3caec707f51408ac2154a" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.", + "work": "Max und Moritz (Wilhelm Busch) — public domain", + "audio": "audio/thorsten.opus", + "sha256": "d45938440534f7fa11afe388fa41a9a56879809a65c8477605220ed53d5a938b" + } + }, + { + "name": "kerstin", + "language": "German", + "language_id": "de", + "gender": "F", + "donor": "kerstin (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "kerstin.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/kerstin.opus", + "sha256": "34d7c0db776bcda98e56bc130ced840debe0b12a73f4b03c556b30fc356db187", + "duration_s": 9.79, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/kerstin.safetensors", + "sha256": "d096ecde83dfc5de7a927acf67a5304c55f863ca722f826c67fafd31c48609a6" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.", + "work": "Max und Moritz (Wilhelm Busch) — public domain", + "audio": "audio/kerstin.opus", + "sha256": "dc1990efd85eafee75810f45fd259505d4840579f008aba9f38ca1a7c42e839b" + } + }, + { + "name": "henri", + "language": "French", + "language_id": "fr", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": "10087", + "source": { + "name": "Kyutai tts-voices, cml-tts french enhanced references", + "url": "https://huggingface.co/kyutai/tts-voices", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "henri.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/henri.opus", + "sha256": "9185f21daf7bcbb9a53ea06b059438ea83b7315ef4f852d69d56afc562061fc0", + "duration_s": 6.56, + "construction": "pre-cut enhanced 10 s reference" + }, + "profile": { + "hf_path": "voices/henri.safetensors", + "sha256": "2be79463e8a2cfe339a7d6088e397c22be4430c6796e76362a0f33f351bf940a" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.", + "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain", + "audio": "audio/henri.opus", + "sha256": "9a63ac8258c23aa750c871b4cc050da5658cd44b77448aad3cf1caef5ecb047b" + } + }, + { + "name": "colette", + "language": "French", + "language_id": "fr", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "1406", + "source": { + "name": "Kyutai tts-voices, cml-tts french enhanced references", + "url": "https://huggingface.co/kyutai/tts-voices", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "colette.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/colette.opus", + "sha256": "324d657ffdc1d1651e9da4ad57258b1c54ee429b437f909ac29130e78cee21a9", + "duration_s": 7.24, + "construction": "pre-cut enhanced 10 s reference" + }, + "profile": { + "hf_path": "voices/colette.safetensors", + "sha256": "bdccafdc54f700cb456e3f531d0f9da3e5cc0c9fcece8284db44773f99c162c9" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.", + "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain", + "audio": "audio/colette.opus", + "sha256": "e9f0084bb45e1b7647eec082dc03f43a110e10f915a2baf99d5776201ec664bd" + } + }, + { + "name": "pim", + "language": "Dutch", + "language_id": "nl", + "gender": "M", + "donor": "pim (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "pim.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/pim.opus", + "sha256": "b7428579a45f688ebf6602fd94e6a92dfe34b20cf34e7aae3c73fb73b2182d29", + "duration_s": 11.0, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/pim.safetensors", + "sha256": "05c5b3bd921240acc39cc8ffd576ab71dcbc36e6238ce468292ae74c61190130" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.", + "work": "De pruimeboom (Hieronymus van Alphen) — public domain", + "audio": "audio/pim.opus", + "sha256": "c00535642d18623f4d2085fa0bde17bdd08c7c8b9b38ade8a85dfee7b9974b37" + } + }, + { + "name": "nathalie", + "language": "Dutch", + "language_id": "nl", + "gender": "F", + "donor": "nathalie (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "nathalie.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/nathalie.opus", + "sha256": "bc09a866b18264d555a35522e3319ff820914f59a8c3dd8687c01ee16203e5fd", + "duration_s": 10.8, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/nathalie.safetensors", + "sha256": "a61158533e125600613719617ccc89ea4b66fbc9fbe149ee02e3b5c4113cc11d" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.", + "work": "De pruimeboom (Hieronymus van Alphen) — public domain", + "audio": "audio/nathalie.opus", + "sha256": "d2bf4970c1cf1d8f4437ee183e67ed5d1f67683249d26e36c6bf7cf3904490da" + } + }, + { + "name": "dave", + "language": "Spanish", + "language_id": "es", + "gender": "M", + "donor": "dave (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "dave.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/dave.opus", + "sha256": "9e30010eafeb37abb20a9976e17bfa3d313f779d5703bba9a536791916b162d6", + "duration_s": 11.0, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/dave.safetensors", + "sha256": "14b5e58a6e5a457b516f5a9ae02d3a739449a02a92d0791b356377b68c366cb4" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.", + "work": "La mona (Felix Maria de Samaniego) — public domain", + "audio": "audio/dave.opus", + "sha256": "c1240b63c1f83b5279dccd028c18410d4051f590e695199a91bdd87af4c08d22" + } + }, + { + "name": "carmen", + "language": "Spanish", + "language_id": "es", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "2308", + "source": { + "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)", + "url": "https://huggingface.co/datasets/ylacombe/cml-tts", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "carmen.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/carmen.opus", + "sha256": "ec9a38e7c13420543a004c4db665af8fb1b66a952f18edeb98d0099a14f186e3", + "duration_s": 14.37, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/carmen.safetensors", + "sha256": "c9387f3bc58e8e1dd41700339e886a546200597fe629e523282ba4dbd4ddc069" + }, + "speaker_similarity": null, + "sample": { + "seed": 3, + "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.", + "work": "La mona (Felix Maria de Samaniego) — public domain", + "audio": "audio/carmen.opus", + "sha256": "aefb626d30e906f729e8ea4c7a178f3b0b3092851fdbb1af2578fb840ae2078c" + } + }, + { + "name": "dante", + "language": "Italian", + "language_id": "it", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": "12598", + "source": { + "name": "Multilingual LibriSpeech (facebook/multilingual_librispeech)", + "url": "https://huggingface.co/datasets/facebook/multilingual_librispeech", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "dante.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/dante.opus", + "sha256": "df142e760a870aaf53d71e945fcafbdfb2c5bcf14676c4c8b90ded229b9a48fe", + "duration_s": 14.0, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/dante.safetensors", + "sha256": "d0ba2cf9871b7f933ff5f0dbdca9578077599c2e8209fc6246727edd79c15924" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.", + "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain", + "audio": "audio/dante.opus", + "sha256": "9b05b0d44b83e208ea4c97b30a4b2e03cb323e5108adf0847f1572f18639fc04" + } + }, + { + "name": "paola", + "language": "Italian", + "language_id": "it", + "gender": "F", + "donor": "paola (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "paola.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/paola.opus", + "sha256": "424e0cecda21fa34c5a05fb5f6bc029949f0485ac91e7fd45afc7645ff4cdbb9", + "duration_s": 11.0, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/paola.safetensors", + "sha256": "6c8bcd3d39dce92955833b1b9b2d483d1f8e48ac48fb772acbb63d45e4e00f6c" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.", + "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain", + "audio": "audio/paola.opus", + "sha256": "452ae96e4a6195a781c1aa521d7a856300ae814541856702824ad413527dabd3" + } + }, + { + "name": "tugao", + "language": "Portuguese (European)", + "language_id": "pt", + "gender": "M", + "donor": "tugao (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "tugao.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/tugao.opus", + "sha256": "d6b40a71decdc9e9a560442f0c0e78f412b4d94365c9bddcb3e215ca4f452118", + "duration_s": 9.78, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/tugao.safetensors", + "sha256": "adcebf6fe42125bdc0aaca6ba3d23d79a30c8ad6a52aa65fd7970bed540eec43" + }, + "speaker_similarity": 0.916, + "sample": { + "seed": 7, + "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.", + "work": "A Cidade e as Serras (Eca de Queiros) — public domain", + "audio": "audio/tugao.opus", + "sha256": "ddf3be70b986247fc879f90d5dc3f9e41e28d763e08ca10de1a44ea08e3f112b" + } + }, + { + "name": "ines", + "language": "Portuguese (European)", + "language_id": "pt", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "7925", + "source": { + "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)", + "url": "https://huggingface.co/datasets/ylacombe/cml-tts", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "ines.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/ines.opus", + "sha256": "59f2fde8e98c8183e995e2576128ac4bbb186c08f677c1ac6daf2cd359bba7f4", + "duration_s": 10.44, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/ines.safetensors", + "sha256": "b7f76b52170dfcdcef0e8835956967bf2fbc4973883b6e99d6b2be7fd4f3a496" + }, + "speaker_similarity": 0.96, + "sample": { + "seed": 7, + "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.", + "work": "A Cidade e as Serras (Eca de Queiros) — public domain", + "audio": "audio/ines.opus", + "sha256": "55c9cb8a3716ccfce280c25b577e344363fe3978b35577ce8c20ee6d4f3869ea" + }, + "notes": "CML-TTS is LibriVox-derived; accent is Brazilian Portuguese, unlike tugao (European)." + }, + { + "name": "nils", + "language": "Swedish", + "language_id": "sv", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": null, + "source": { + "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)", + "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/", + "license": "CC0-1.0", + "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "nils.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/nils.opus", + "sha256": "26680c82a976fe2d6f3202a97b50499274f1f0c36d329cbe6cdd6c3ebac1f0f0", + "duration_s": 11.0, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/nils.safetensors", + "sha256": "be75721c99547ab3ecdf1f8629101e9fa3ea1d569fc6e1220b8e2f259fb49f4e" + }, + "speaker_similarity": 0.92, + "sample": { + "seed": 7, + "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.", + "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain", + "audio": "audio/nils.opus", + "sha256": "61b704118c82878a218d0db2b0fd3da20a70d0d442e2a36cc064ccc1a7f0679b" + } + }, + { + "name": "selma", + "language": "Swedish", + "language_id": "sv", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": null, + "source": { + "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)", + "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/", + "license": "CC0-1.0", + "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "selma.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/selma.opus", + "sha256": "775d4db5a95ac9fea95b1f9b8d6a5b1917a3aefd466e153962ad195d83ed51d0", + "duration_s": 8.53, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/selma.safetensors", + "sha256": "d3ba9ab18c676db23028c9561ad79a183b35adae377d20d46f1d8f253ce6eeec" + }, + "speaker_similarity": 0.925, + "sample": { + "seed": 7, + "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.", + "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain", + "audio": "audio/selma.opus", + "sha256": "70f186995d8b6aac965662cdff346ec5211bef98276c7ef94a69c096dccd0782" + } + }, + { + "name": "soren", + "language": "Danish", + "language_id": "da", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": "37", + "source": { + "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)", + "url": "https://huggingface.co/datasets/alexandrainst/nst-da", + "license": "CC0-1.0", + "consent": "corpus released CC0; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "soren.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/soren.opus", + "sha256": "a0d7510fa2de66b8bf656c5f2ff93ddf462878ce31799860ca242bcad64487fb", + "duration_s": 7.38, + "construction": "single continuous clip, close mic" + }, + "profile": { + "hf_path": "voices/soren.safetensors", + "sha256": "30d9fe9ee02928cd65ef7f4cb286c401c1809f07c13c19de79ffa375a027412e" + }, + "speaker_similarity": 0.893, + "sample": { + "seed": 7, + "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.", + "work": "Den grimme aelling (H. C. Andersen) — public domain", + "audio": "audio/soren.opus", + "sha256": "f9841dd5162d30df3e67aee92128bd3708f4eff6c3d1c5c177091765cf04af2c" + } + }, + { + "name": "freja", + "language": "Danish", + "language_id": "da", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "35", + "source": { + "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)", + "url": "https://huggingface.co/datasets/alexandrainst/nst-da", + "license": "CC0-1.0", + "consent": "corpus released CC0; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "freja.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/freja.opus", + "sha256": "078baf737c9acc6f567dd87a8b3eca897becb0e83d9209279bf8314830d2678a", + "duration_s": 8.78, + "construction": "single continuous clip, close mic" + }, + "profile": { + "hf_path": "voices/freja.safetensors", + "sha256": "0a3f913b9b47c529393f843850228bd8eee6a883b4c138978fd3afa7758561e8" + }, + "speaker_similarity": 0.907, + "sample": { + "seed": 7, + "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.", + "work": "Den grimme aelling (H. C. Andersen) — public domain", + "audio": "audio/freja.opus", + "sha256": "cd785cc844250a5ae97010e2a3b0ee9260c5a24d016350d931aa7ec776958812" + } + } +] diff --git a/hf-loudkit/requirements.txt b/hf-loudkit/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..60b4bdb741c3b04aa1bb164e82cbfb4bbb071a24 --- /dev/null +++ b/hf-loudkit/requirements.txt @@ -0,0 +1,18 @@ +# The library under demo. 0.1.0 is on PyPI, so this installs from the registry +# rather than carrying a wheel beside the app. +# +# torch the cuda backend this Space runs on +# audio soundfile, for Result.save +# enroll torchaudio + librosa, for the Clone tab +# hub huggingface_hub, to resolve loudreader/loudr-1 +loudkit[torch,audio,enroll,hub]==0.1.0 + +# loudkit floors torch at >=2.4 and sets no ceiling, and ZeroGPU runs a fixed +# set of builds: 2.8.0, 2.9.1, 2.10.0, 2.11.0. Pinned here so the resolver +# cannot land outside that window. torchaudio tracks torch version for version. +torch==2.8.0 +torchaudio==2.8.0 + +# The ZeroGPU decorator. Present on ZeroGPU hardware already; named here so a +# local run or a duplicated Space installs it too. +spaces>=0.42 diff --git a/hf-loudkit/voices.json b/hf-loudkit/voices.json new file mode 100644 index 0000000000000000000000000000000000000000..ac89f1f9159d32cd3c35686f32448d57a27284b2 --- /dev/null +++ b/hf-loudkit/voices.json @@ -0,0 +1,683 @@ +[ + { + "name": "darkman", + "language": "Polish", + "language_id": "pl", + "gender": "M", + "donor": "darkman (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "darkman.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/darkman.opus", + "sha256": "d228db9d17b1ddb4ba450f16a6b8650bb0a8550521b78a914cb9e573cd24d576", + "duration_s": 9.78, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/darkman.safetensors", + "sha256": "78988a6769f90c209240f9f285eb9cdd760d1ba471cfa7d6e67db632599cfa1b" + }, + "speaker_similarity": 0.935, + "sample": { + "seed": 7, + "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.", + "work": "Quo Vadis (Henryk Sienkiewicz) — public domain", + "audio": "audio/darkman.opus", + "sha256": "b30f98cf0452e9c694fcbafc27b413041c87022238f5424fc1aa452c8eb156a5" + } + }, + { + "name": "gosia", + "language": "Polish", + "language_id": "pl", + "gender": "F", + "donor": "gosia (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "gosia.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/gosia.opus", + "sha256": "02c1d09db8b097c66dd8705747a6ef138596f5f91c3213b1cd430ae75ca2279a", + "duration_s": 10.86, + "construction": "concatenated long donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/gosia.safetensors", + "sha256": "e115cf64bfa0ad15770a50c0be8097312a58d41f84c7bb4f01ecbdbac5d21823" + }, + "speaker_similarity": 0.917, + "sample": { + "seed": 7, + "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.", + "work": "Quo Vadis (Henryk Sienkiewicz) — public domain", + "audio": "audio/gosia.opus", + "sha256": "4ba1107d7f807ac2e0941167d1bc4a7cfb2039a4b0482a3bc798db5eeefe17e5" + } + }, + { + "name": "joe", + "language": "English", + "language_id": "en", + "gender": "M", + "donor": "joe (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "joe.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/joe.opus", + "sha256": "591c8d6a5799a4266d029762dc0f671ac9f0455382aeef1e763ca206d6fdd164", + "duration_s": 10.99, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/joe.safetensors", + "sha256": "cfc3687fdb1f6b56fe13e252ca4a5bd681cef0702dc5cd56d4ebbc60e1aea6e8" + }, + "speaker_similarity": 0.924, + "sample": { + "seed": 11, + "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.", + "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain", + "audio": "audio/joe.opus", + "sha256": "83e0bf0a1f6b03e047de66871536c3ff77291041d0a3f2907b6990e580519fea" + } + }, + { + "name": "kathleen", + "language": "English", + "language_id": "en", + "gender": "F", + "donor": "kathleen (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "kathleen.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/kathleen.opus", + "sha256": "f43a24359a333e1f723c5f10d7dc799ec5880abf2b5b9321a7d5f8c7064d6f73", + "duration_s": 11.0, + "construction": "concatenated donation clips; prompt texts from the CMU ARCTIC prompt list (public-domain Gutenberg novels), recording is kathleen's own CC0 donation" + }, + "profile": { + "hf_path": "voices/kathleen.safetensors", + "sha256": "6a845e4eb11989fc9bd8e19eed7c5ac585e6d08e1ec824211f0eb240ce523d29" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.", + "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain", + "audio": "audio/kathleen.opus", + "sha256": "48615517c1d1d2d55dbb7f74f37982918236e33f079947459d56c116cd70b7a3" + } + }, + { + "name": "thorsten", + "language": "German", + "language_id": "de", + "gender": "M", + "donor": "Thorsten Mueller", + "speaker_id": null, + "source": { + "name": "Thorsten-Voice TV-44kHz-Full", + "url": "https://huggingface.co/datasets/Thorsten-Voice/TV-44kHz-Full", + "license": "CC0-1.0", + "consent": "voice deliberately donated by Thorsten Mueller for TTS" + }, + "reference": { + "source_filename": "thorsten.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/thorsten.opus", + "sha256": "37805d1136f727f7530e714b772c23224ae1c6f6931ea5b78399cf286694de3e", + "duration_s": 6.37, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/thorsten.safetensors", + "sha256": "87873fec8d3dceaa643ab1ab07cf5143364e659dd6b3caec707f51408ac2154a" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.", + "work": "Max und Moritz (Wilhelm Busch) — public domain", + "audio": "audio/thorsten.opus", + "sha256": "d45938440534f7fa11afe388fa41a9a56879809a65c8477605220ed53d5a938b" + } + }, + { + "name": "kerstin", + "language": "German", + "language_id": "de", + "gender": "F", + "donor": "kerstin (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "kerstin.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/kerstin.opus", + "sha256": "34d7c0db776bcda98e56bc130ced840debe0b12a73f4b03c556b30fc356db187", + "duration_s": 9.79, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/kerstin.safetensors", + "sha256": "d096ecde83dfc5de7a927acf67a5304c55f863ca722f826c67fafd31c48609a6" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.", + "work": "Max und Moritz (Wilhelm Busch) — public domain", + "audio": "audio/kerstin.opus", + "sha256": "dc1990efd85eafee75810f45fd259505d4840579f008aba9f38ca1a7c42e839b" + } + }, + { + "name": "henri", + "language": "French", + "language_id": "fr", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": "10087", + "source": { + "name": "Kyutai tts-voices, cml-tts french enhanced references", + "url": "https://huggingface.co/kyutai/tts-voices", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "henri.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/henri.opus", + "sha256": "9185f21daf7bcbb9a53ea06b059438ea83b7315ef4f852d69d56afc562061fc0", + "duration_s": 6.56, + "construction": "pre-cut enhanced 10 s reference" + }, + "profile": { + "hf_path": "voices/henri.safetensors", + "sha256": "2be79463e8a2cfe339a7d6088e397c22be4430c6796e76362a0f33f351bf940a" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.", + "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain", + "audio": "audio/henri.opus", + "sha256": "9a63ac8258c23aa750c871b4cc050da5658cd44b77448aad3cf1caef5ecb047b" + } + }, + { + "name": "colette", + "language": "French", + "language_id": "fr", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "1406", + "source": { + "name": "Kyutai tts-voices, cml-tts french enhanced references", + "url": "https://huggingface.co/kyutai/tts-voices", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "colette.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/colette.opus", + "sha256": "324d657ffdc1d1651e9da4ad57258b1c54ee429b437f909ac29130e78cee21a9", + "duration_s": 7.24, + "construction": "pre-cut enhanced 10 s reference" + }, + "profile": { + "hf_path": "voices/colette.safetensors", + "sha256": "bdccafdc54f700cb456e3f531d0f9da3e5cc0c9fcece8284db44773f99c162c9" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.", + "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain", + "audio": "audio/colette.opus", + "sha256": "e9f0084bb45e1b7647eec082dc03f43a110e10f915a2baf99d5776201ec664bd" + } + }, + { + "name": "pim", + "language": "Dutch", + "language_id": "nl", + "gender": "M", + "donor": "pim (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "pim.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/pim.opus", + "sha256": "b7428579a45f688ebf6602fd94e6a92dfe34b20cf34e7aae3c73fb73b2182d29", + "duration_s": 11.0, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/pim.safetensors", + "sha256": "05c5b3bd921240acc39cc8ffd576ab71dcbc36e6238ce468292ae74c61190130" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.", + "work": "De pruimeboom (Hieronymus van Alphen) — public domain", + "audio": "audio/pim.opus", + "sha256": "c00535642d18623f4d2085fa0bde17bdd08c7c8b9b38ade8a85dfee7b9974b37" + } + }, + { + "name": "nathalie", + "language": "Dutch", + "language_id": "nl", + "gender": "F", + "donor": "nathalie (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "nathalie.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/nathalie.opus", + "sha256": "bc09a866b18264d555a35522e3319ff820914f59a8c3dd8687c01ee16203e5fd", + "duration_s": 10.8, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/nathalie.safetensors", + "sha256": "a61158533e125600613719617ccc89ea4b66fbc9fbe149ee02e3b5c4113cc11d" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.", + "work": "De pruimeboom (Hieronymus van Alphen) — public domain", + "audio": "audio/nathalie.opus", + "sha256": "d2bf4970c1cf1d8f4437ee183e67ed5d1f67683249d26e36c6bf7cf3904490da" + } + }, + { + "name": "dave", + "language": "Spanish", + "language_id": "es", + "gender": "M", + "donor": "dave (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "dave.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/dave.opus", + "sha256": "9e30010eafeb37abb20a9976e17bfa3d313f779d5703bba9a536791916b162d6", + "duration_s": 11.0, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/dave.safetensors", + "sha256": "14b5e58a6e5a457b516f5a9ae02d3a739449a02a92d0791b356377b68c366cb4" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.", + "work": "La mona (Felix Maria de Samaniego) — public domain", + "audio": "audio/dave.opus", + "sha256": "c1240b63c1f83b5279dccd028c18410d4051f590e695199a91bdd87af4c08d22" + } + }, + { + "name": "carmen", + "language": "Spanish", + "language_id": "es", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "2308", + "source": { + "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)", + "url": "https://huggingface.co/datasets/ylacombe/cml-tts", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "carmen.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/carmen.opus", + "sha256": "ec9a38e7c13420543a004c4db665af8fb1b66a952f18edeb98d0099a14f186e3", + "duration_s": 14.37, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/carmen.safetensors", + "sha256": "c9387f3bc58e8e1dd41700339e886a546200597fe629e523282ba4dbd4ddc069" + }, + "speaker_similarity": null, + "sample": { + "seed": 3, + "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.", + "work": "La mona (Felix Maria de Samaniego) — public domain", + "audio": "audio/carmen.opus", + "sha256": "aefb626d30e906f729e8ea4c7a178f3b0b3092851fdbb1af2578fb840ae2078c" + } + }, + { + "name": "dante", + "language": "Italian", + "language_id": "it", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": "12598", + "source": { + "name": "Multilingual LibriSpeech (facebook/multilingual_librispeech)", + "url": "https://huggingface.co/datasets/facebook/multilingual_librispeech", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "dante.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/dante.opus", + "sha256": "df142e760a870aaf53d71e945fcafbdfb2c5bcf14676c4c8b90ded229b9a48fe", + "duration_s": 14.0, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/dante.safetensors", + "sha256": "d0ba2cf9871b7f933ff5f0dbdca9578077599c2e8209fc6246727edd79c15924" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.", + "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain", + "audio": "audio/dante.opus", + "sha256": "9b05b0d44b83e208ea4c97b30a4b2e03cb323e5108adf0847f1572f18639fc04" + } + }, + { + "name": "paola", + "language": "Italian", + "language_id": "it", + "gender": "F", + "donor": "paola (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "paola.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/paola.opus", + "sha256": "424e0cecda21fa34c5a05fb5f6bc029949f0485ac91e7fd45afc7645ff4cdbb9", + "duration_s": 11.0, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/paola.safetensors", + "sha256": "6c8bcd3d39dce92955833b1b9b2d483d1f8e48ac48fb772acbb63d45e4e00f6c" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.", + "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain", + "audio": "audio/paola.opus", + "sha256": "452ae96e4a6195a781c1aa521d7a856300ae814541856702824ad413527dabd3" + } + }, + { + "name": "tugao", + "language": "Portuguese (European)", + "language_id": "pt", + "gender": "M", + "donor": "tugao (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "tugao.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/tugao.opus", + "sha256": "d6b40a71decdc9e9a560442f0c0e78f412b4d94365c9bddcb3e215ca4f452118", + "duration_s": 9.78, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/tugao.safetensors", + "sha256": "adcebf6fe42125bdc0aaca6ba3d23d79a30c8ad6a52aa65fd7970bed540eec43" + }, + "speaker_similarity": 0.916, + "sample": { + "seed": 7, + "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.", + "work": "A Cidade e as Serras (Eca de Queiros) — public domain", + "audio": "audio/tugao.opus", + "sha256": "ddf3be70b986247fc879f90d5dc3f9e41e28d763e08ca10de1a44ea08e3f112b" + } + }, + { + "name": "ines", + "language": "Portuguese (European)", + "language_id": "pt", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "7925", + "source": { + "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)", + "url": "https://huggingface.co/datasets/ylacombe/cml-tts", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "ines.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/ines.opus", + "sha256": "59f2fde8e98c8183e995e2576128ac4bbb186c08f677c1ac6daf2cd359bba7f4", + "duration_s": 10.44, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/ines.safetensors", + "sha256": "b7f76b52170dfcdcef0e8835956967bf2fbc4973883b6e99d6b2be7fd4f3a496" + }, + "speaker_similarity": 0.96, + "sample": { + "seed": 7, + "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.", + "work": "A Cidade e as Serras (Eca de Queiros) — public domain", + "audio": "audio/ines.opus", + "sha256": "55c9cb8a3716ccfce280c25b577e344363fe3978b35577ce8c20ee6d4f3869ea" + }, + "notes": "CML-TTS is LibriVox-derived; accent is Brazilian Portuguese, unlike tugao (European)." + }, + { + "name": "nils", + "language": "Swedish", + "language_id": "sv", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": null, + "source": { + "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)", + "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/", + "license": "CC0-1.0", + "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "nils.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/nils.opus", + "sha256": "26680c82a976fe2d6f3202a97b50499274f1f0c36d329cbe6cdd6c3ebac1f0f0", + "duration_s": 11.0, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/nils.safetensors", + "sha256": "be75721c99547ab3ecdf1f8629101e9fa3ea1d569fc6e1220b8e2f259fb49f4e" + }, + "speaker_similarity": 0.92, + "sample": { + "seed": 7, + "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.", + "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain", + "audio": "audio/nils.opus", + "sha256": "61b704118c82878a218d0db2b0fd3da20a70d0d442e2a36cc064ccc1a7f0679b" + } + }, + { + "name": "selma", + "language": "Swedish", + "language_id": "sv", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": null, + "source": { + "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)", + "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/", + "license": "CC0-1.0", + "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "selma.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/selma.opus", + "sha256": "775d4db5a95ac9fea95b1f9b8d6a5b1917a3aefd466e153962ad195d83ed51d0", + "duration_s": 8.53, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/selma.safetensors", + "sha256": "d3ba9ab18c676db23028c9561ad79a183b35adae377d20d46f1d8f253ce6eeec" + }, + "speaker_similarity": 0.925, + "sample": { + "seed": 7, + "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.", + "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain", + "audio": "audio/selma.opus", + "sha256": "70f186995d8b6aac965662cdff346ec5211bef98276c7ef94a69c096dccd0782" + } + }, + { + "name": "soren", + "language": "Danish", + "language_id": "da", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": "37", + "source": { + "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)", + "url": "https://huggingface.co/datasets/alexandrainst/nst-da", + "license": "CC0-1.0", + "consent": "corpus released CC0; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "soren.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/soren.opus", + "sha256": "a0d7510fa2de66b8bf656c5f2ff93ddf462878ce31799860ca242bcad64487fb", + "duration_s": 7.38, + "construction": "single continuous clip, close mic" + }, + "profile": { + "hf_path": "voices/soren.safetensors", + "sha256": "30d9fe9ee02928cd65ef7f4cb286c401c1809f07c13c19de79ffa375a027412e" + }, + "speaker_similarity": 0.893, + "sample": { + "seed": 7, + "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.", + "work": "Den grimme aelling (H. C. Andersen) — public domain", + "audio": "audio/soren.opus", + "sha256": "f9841dd5162d30df3e67aee92128bd3708f4eff6c3d1c5c177091765cf04af2c" + } + }, + { + "name": "freja", + "language": "Danish", + "language_id": "da", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "35", + "source": { + "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)", + "url": "https://huggingface.co/datasets/alexandrainst/nst-da", + "license": "CC0-1.0", + "consent": "corpus released CC0; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "freja.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/freja.opus", + "sha256": "078baf737c9acc6f567dd87a8b3eca897becb0e83d9209279bf8314830d2678a", + "duration_s": 8.78, + "construction": "single continuous clip, close mic" + }, + "profile": { + "hf_path": "voices/freja.safetensors", + "sha256": "0a3f913b9b47c529393f843850228bd8eee6a883b4c138978fd3afa7758561e8" + }, + "speaker_similarity": 0.907, + "sample": { + "seed": 7, + "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.", + "work": "Den grimme aelling (H. C. Andersen) — public domain", + "audio": "audio/freja.opus", + "sha256": "cd785cc844250a5ae97010e2a3b0ee9260c5a24d016350d931aa7ec776958812" + } + } +] diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..60b4bdb741c3b04aa1bb164e82cbfb4bbb071a24 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,18 @@ +# The library under demo. 0.1.0 is on PyPI, so this installs from the registry +# rather than carrying a wheel beside the app. +# +# torch the cuda backend this Space runs on +# audio soundfile, for Result.save +# enroll torchaudio + librosa, for the Clone tab +# hub huggingface_hub, to resolve loudreader/loudr-1 +loudkit[torch,audio,enroll,hub]==0.1.0 + +# loudkit floors torch at >=2.4 and sets no ceiling, and ZeroGPU runs a fixed +# set of builds: 2.8.0, 2.9.1, 2.10.0, 2.11.0. Pinned here so the resolver +# cannot land outside that window. torchaudio tracks torch version for version. +torch==2.8.0 +torchaudio==2.8.0 + +# The ZeroGPU decorator. Present on ZeroGPU hardware already; named here so a +# local run or a duplicated Space installs it too. +spaces>=0.42 diff --git a/voices.json b/voices.json new file mode 100644 index 0000000000000000000000000000000000000000..ac89f1f9159d32cd3c35686f32448d57a27284b2 --- /dev/null +++ b/voices.json @@ -0,0 +1,683 @@ +[ + { + "name": "darkman", + "language": "Polish", + "language_id": "pl", + "gender": "M", + "donor": "darkman (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "darkman.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/darkman.opus", + "sha256": "d228db9d17b1ddb4ba450f16a6b8650bb0a8550521b78a914cb9e573cd24d576", + "duration_s": 9.78, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/darkman.safetensors", + "sha256": "78988a6769f90c209240f9f285eb9cdd760d1ba471cfa7d6e67db632599cfa1b" + }, + "speaker_similarity": 0.935, + "sample": { + "seed": 7, + "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.", + "work": "Quo Vadis (Henryk Sienkiewicz) — public domain", + "audio": "audio/darkman.opus", + "sha256": "b30f98cf0452e9c694fcbafc27b413041c87022238f5424fc1aa452c8eb156a5" + } + }, + { + "name": "gosia", + "language": "Polish", + "language_id": "pl", + "gender": "F", + "donor": "gosia (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "gosia.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/gosia.opus", + "sha256": "02c1d09db8b097c66dd8705747a6ef138596f5f91c3213b1cd430ae75ca2279a", + "duration_s": 10.86, + "construction": "concatenated long donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/gosia.safetensors", + "sha256": "e115cf64bfa0ad15770a50c0be8097312a58d41f84c7bb4f01ecbdbac5d21823" + }, + "speaker_similarity": 0.917, + "sample": { + "seed": 7, + "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.", + "work": "Quo Vadis (Henryk Sienkiewicz) — public domain", + "audio": "audio/gosia.opus", + "sha256": "4ba1107d7f807ac2e0941167d1bc4a7cfb2039a4b0482a3bc798db5eeefe17e5" + } + }, + { + "name": "joe", + "language": "English", + "language_id": "en", + "gender": "M", + "donor": "joe (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "joe.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/joe.opus", + "sha256": "591c8d6a5799a4266d029762dc0f671ac9f0455382aeef1e763ca206d6fdd164", + "duration_s": 10.99, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/joe.safetensors", + "sha256": "cfc3687fdb1f6b56fe13e252ca4a5bd681cef0702dc5cd56d4ebbc60e1aea6e8" + }, + "speaker_similarity": 0.924, + "sample": { + "seed": 11, + "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.", + "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain", + "audio": "audio/joe.opus", + "sha256": "83e0bf0a1f6b03e047de66871536c3ff77291041d0a3f2907b6990e580519fea" + } + }, + { + "name": "kathleen", + "language": "English", + "language_id": "en", + "gender": "F", + "donor": "kathleen (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "kathleen.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/kathleen.opus", + "sha256": "f43a24359a333e1f723c5f10d7dc799ec5880abf2b5b9321a7d5f8c7064d6f73", + "duration_s": 11.0, + "construction": "concatenated donation clips; prompt texts from the CMU ARCTIC prompt list (public-domain Gutenberg novels), recording is kathleen's own CC0 donation" + }, + "profile": { + "hf_path": "voices/kathleen.safetensors", + "sha256": "6a845e4eb11989fc9bd8e19eed7c5ac585e6d08e1ec824211f0eb240ce523d29" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.", + "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain", + "audio": "audio/kathleen.opus", + "sha256": "48615517c1d1d2d55dbb7f74f37982918236e33f079947459d56c116cd70b7a3" + } + }, + { + "name": "thorsten", + "language": "German", + "language_id": "de", + "gender": "M", + "donor": "Thorsten Mueller", + "speaker_id": null, + "source": { + "name": "Thorsten-Voice TV-44kHz-Full", + "url": "https://huggingface.co/datasets/Thorsten-Voice/TV-44kHz-Full", + "license": "CC0-1.0", + "consent": "voice deliberately donated by Thorsten Mueller for TTS" + }, + "reference": { + "source_filename": "thorsten.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/thorsten.opus", + "sha256": "37805d1136f727f7530e714b772c23224ae1c6f6931ea5b78399cf286694de3e", + "duration_s": 6.37, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/thorsten.safetensors", + "sha256": "87873fec8d3dceaa643ab1ab07cf5143364e659dd6b3caec707f51408ac2154a" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.", + "work": "Max und Moritz (Wilhelm Busch) — public domain", + "audio": "audio/thorsten.opus", + "sha256": "d45938440534f7fa11afe388fa41a9a56879809a65c8477605220ed53d5a938b" + } + }, + { + "name": "kerstin", + "language": "German", + "language_id": "de", + "gender": "F", + "donor": "kerstin (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "kerstin.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/kerstin.opus", + "sha256": "34d7c0db776bcda98e56bc130ced840debe0b12a73f4b03c556b30fc356db187", + "duration_s": 9.79, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/kerstin.safetensors", + "sha256": "d096ecde83dfc5de7a927acf67a5304c55f863ca722f826c67fafd31c48609a6" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.", + "work": "Max und Moritz (Wilhelm Busch) — public domain", + "audio": "audio/kerstin.opus", + "sha256": "dc1990efd85eafee75810f45fd259505d4840579f008aba9f38ca1a7c42e839b" + } + }, + { + "name": "henri", + "language": "French", + "language_id": "fr", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": "10087", + "source": { + "name": "Kyutai tts-voices, cml-tts french enhanced references", + "url": "https://huggingface.co/kyutai/tts-voices", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "henri.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/henri.opus", + "sha256": "9185f21daf7bcbb9a53ea06b059438ea83b7315ef4f852d69d56afc562061fc0", + "duration_s": 6.56, + "construction": "pre-cut enhanced 10 s reference" + }, + "profile": { + "hf_path": "voices/henri.safetensors", + "sha256": "2be79463e8a2cfe339a7d6088e397c22be4430c6796e76362a0f33f351bf940a" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.", + "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain", + "audio": "audio/henri.opus", + "sha256": "9a63ac8258c23aa750c871b4cc050da5658cd44b77448aad3cf1caef5ecb047b" + } + }, + { + "name": "colette", + "language": "French", + "language_id": "fr", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "1406", + "source": { + "name": "Kyutai tts-voices, cml-tts french enhanced references", + "url": "https://huggingface.co/kyutai/tts-voices", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "colette.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/colette.opus", + "sha256": "324d657ffdc1d1651e9da4ad57258b1c54ee429b437f909ac29130e78cee21a9", + "duration_s": 7.24, + "construction": "pre-cut enhanced 10 s reference" + }, + "profile": { + "hf_path": "voices/colette.safetensors", + "sha256": "bdccafdc54f700cb456e3f531d0f9da3e5cc0c9fcece8284db44773f99c162c9" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.", + "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain", + "audio": "audio/colette.opus", + "sha256": "e9f0084bb45e1b7647eec082dc03f43a110e10f915a2baf99d5776201ec664bd" + } + }, + { + "name": "pim", + "language": "Dutch", + "language_id": "nl", + "gender": "M", + "donor": "pim (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "pim.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/pim.opus", + "sha256": "b7428579a45f688ebf6602fd94e6a92dfe34b20cf34e7aae3c73fb73b2182d29", + "duration_s": 11.0, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/pim.safetensors", + "sha256": "05c5b3bd921240acc39cc8ffd576ab71dcbc36e6238ce468292ae74c61190130" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.", + "work": "De pruimeboom (Hieronymus van Alphen) — public domain", + "audio": "audio/pim.opus", + "sha256": "c00535642d18623f4d2085fa0bde17bdd08c7c8b9b38ade8a85dfee7b9974b37" + } + }, + { + "name": "nathalie", + "language": "Dutch", + "language_id": "nl", + "gender": "F", + "donor": "nathalie (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "nathalie.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/nathalie.opus", + "sha256": "bc09a866b18264d555a35522e3319ff820914f59a8c3dd8687c01ee16203e5fd", + "duration_s": 10.8, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/nathalie.safetensors", + "sha256": "a61158533e125600613719617ccc89ea4b66fbc9fbe149ee02e3b5c4113cc11d" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.", + "work": "De pruimeboom (Hieronymus van Alphen) — public domain", + "audio": "audio/nathalie.opus", + "sha256": "d2bf4970c1cf1d8f4437ee183e67ed5d1f67683249d26e36c6bf7cf3904490da" + } + }, + { + "name": "dave", + "language": "Spanish", + "language_id": "es", + "gender": "M", + "donor": "dave (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "dave.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/dave.opus", + "sha256": "9e30010eafeb37abb20a9976e17bfa3d313f779d5703bba9a536791916b162d6", + "duration_s": 11.0, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/dave.safetensors", + "sha256": "14b5e58a6e5a457b516f5a9ae02d3a739449a02a92d0791b356377b68c366cb4" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.", + "work": "La mona (Felix Maria de Samaniego) — public domain", + "audio": "audio/dave.opus", + "sha256": "c1240b63c1f83b5279dccd028c18410d4051f590e695199a91bdd87af4c08d22" + } + }, + { + "name": "carmen", + "language": "Spanish", + "language_id": "es", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "2308", + "source": { + "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)", + "url": "https://huggingface.co/datasets/ylacombe/cml-tts", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "carmen.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/carmen.opus", + "sha256": "ec9a38e7c13420543a004c4db665af8fb1b66a952f18edeb98d0099a14f186e3", + "duration_s": 14.37, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/carmen.safetensors", + "sha256": "c9387f3bc58e8e1dd41700339e886a546200597fe629e523282ba4dbd4ddc069" + }, + "speaker_similarity": null, + "sample": { + "seed": 3, + "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.", + "work": "La mona (Felix Maria de Samaniego) — public domain", + "audio": "audio/carmen.opus", + "sha256": "aefb626d30e906f729e8ea4c7a178f3b0b3092851fdbb1af2578fb840ae2078c" + } + }, + { + "name": "dante", + "language": "Italian", + "language_id": "it", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": "12598", + "source": { + "name": "Multilingual LibriSpeech (facebook/multilingual_librispeech)", + "url": "https://huggingface.co/datasets/facebook/multilingual_librispeech", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "dante.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/dante.opus", + "sha256": "df142e760a870aaf53d71e945fcafbdfb2c5bcf14676c4c8b90ded229b9a48fe", + "duration_s": 14.0, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/dante.safetensors", + "sha256": "d0ba2cf9871b7f933ff5f0dbdca9578077599c2e8209fc6246727edd79c15924" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.", + "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain", + "audio": "audio/dante.opus", + "sha256": "9b05b0d44b83e208ea4c97b30a4b2e03cb323e5108adf0847f1572f18639fc04" + } + }, + { + "name": "paola", + "language": "Italian", + "language_id": "it", + "gender": "F", + "donor": "paola (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "paola.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/paola.opus", + "sha256": "424e0cecda21fa34c5a05fb5f6bc029949f0485ac91e7fd45afc7645ff4cdbb9", + "duration_s": 11.0, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/paola.safetensors", + "sha256": "6c8bcd3d39dce92955833b1b9b2d483d1f8e48ac48fb772acbb63d45e4e00f6c" + }, + "speaker_similarity": null, + "sample": { + "seed": 7, + "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.", + "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain", + "audio": "audio/paola.opus", + "sha256": "452ae96e4a6195a781c1aa521d7a856300ae814541856702824ad413527dabd3" + } + }, + { + "name": "tugao", + "language": "Portuguese (European)", + "language_id": "pt", + "gender": "M", + "donor": "tugao (donor alias)", + "speaker_id": null, + "source": { + "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)", + "url": "https://github.com/NabuCasa/voice-datasets", + "license": "CC0-1.0", + "consent": "recorded and donated expressly for building TTS voices" + }, + "reference": { + "source_filename": "tugao.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/tugao.opus", + "sha256": "d6b40a71decdc9e9a560442f0c0e78f412b4d94365c9bddcb3e215ca4f452118", + "duration_s": 9.78, + "construction": "concatenated donation clips, 120 ms gaps" + }, + "profile": { + "hf_path": "voices/tugao.safetensors", + "sha256": "adcebf6fe42125bdc0aaca6ba3d23d79a30c8ad6a52aa65fd7970bed540eec43" + }, + "speaker_similarity": 0.916, + "sample": { + "seed": 7, + "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.", + "work": "A Cidade e as Serras (Eca de Queiros) — public domain", + "audio": "audio/tugao.opus", + "sha256": "ddf3be70b986247fc879f90d5dc3f9e41e28d763e08ca10de1a44ea08e3f112b" + } + }, + { + "name": "ines", + "language": "Portuguese (European)", + "language_id": "pt", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "7925", + "source": { + "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)", + "url": "https://huggingface.co/datasets/ylacombe/cml-tts", + "license": "CC-BY-4.0", + "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "ines.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/ines.opus", + "sha256": "59f2fde8e98c8183e995e2576128ac4bbb186c08f677c1ac6daf2cd359bba7f4", + "duration_s": 10.44, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/ines.safetensors", + "sha256": "b7f76b52170dfcdcef0e8835956967bf2fbc4973883b6e99d6b2be7fd4f3a496" + }, + "speaker_similarity": 0.96, + "sample": { + "seed": 7, + "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.", + "work": "A Cidade e as Serras (Eca de Queiros) — public domain", + "audio": "audio/ines.opus", + "sha256": "55c9cb8a3716ccfce280c25b577e344363fe3978b35577ce8c20ee6d4f3869ea" + }, + "notes": "CML-TTS is LibriVox-derived; accent is Brazilian Portuguese, unlike tugao (European)." + }, + { + "name": "nils", + "language": "Swedish", + "language_id": "sv", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": null, + "source": { + "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)", + "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/", + "license": "CC0-1.0", + "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "nils.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/nils.opus", + "sha256": "26680c82a976fe2d6f3202a97b50499274f1f0c36d329cbe6cdd6c3ebac1f0f0", + "duration_s": 11.0, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/nils.safetensors", + "sha256": "be75721c99547ab3ecdf1f8629101e9fa3ea1d569fc6e1220b8e2f259fb49f4e" + }, + "speaker_similarity": 0.92, + "sample": { + "seed": 7, + "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.", + "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain", + "audio": "audio/nils.opus", + "sha256": "61b704118c82878a218d0db2b0fd3da20a70d0d442e2a36cc064ccc1a7f0679b" + } + }, + { + "name": "selma", + "language": "Swedish", + "language_id": "sv", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": null, + "source": { + "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)", + "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/", + "license": "CC0-1.0", + "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "selma.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/selma.opus", + "sha256": "775d4db5a95ac9fea95b1f9b8d6a5b1917a3aefd466e153962ad195d83ed51d0", + "duration_s": 8.53, + "construction": "single continuous clip" + }, + "profile": { + "hf_path": "voices/selma.safetensors", + "sha256": "d3ba9ab18c676db23028c9561ad79a183b35adae377d20d46f1d8f253ce6eeec" + }, + "speaker_similarity": 0.925, + "sample": { + "seed": 7, + "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.", + "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain", + "audio": "audio/selma.opus", + "sha256": "70f186995d8b6aac965662cdff346ec5211bef98276c7ef94a69c096dccd0782" + } + }, + { + "name": "soren", + "language": "Danish", + "language_id": "da", + "gender": "M", + "donor": "anonymous (invented name)", + "speaker_id": "37", + "source": { + "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)", + "url": "https://huggingface.co/datasets/alexandrainst/nst-da", + "license": "CC0-1.0", + "consent": "corpus released CC0; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "soren.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/soren.opus", + "sha256": "a0d7510fa2de66b8bf656c5f2ff93ddf462878ce31799860ca242bcad64487fb", + "duration_s": 7.38, + "construction": "single continuous clip, close mic" + }, + "profile": { + "hf_path": "voices/soren.safetensors", + "sha256": "30d9fe9ee02928cd65ef7f4cb286c401c1809f07c13c19de79ffa375a027412e" + }, + "speaker_similarity": 0.893, + "sample": { + "seed": 7, + "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.", + "work": "Den grimme aelling (H. C. Andersen) — public domain", + "audio": "audio/soren.opus", + "sha256": "f9841dd5162d30df3e67aee92128bd3708f4eff6c3d1c5c177091765cf04af2c" + } + }, + { + "name": "freja", + "language": "Danish", + "language_id": "da", + "gender": "F", + "donor": "anonymous (invented name)", + "speaker_id": "35", + "source": { + "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)", + "url": "https://huggingface.co/datasets/alexandrainst/nst-da", + "license": "CC0-1.0", + "consent": "corpus released CC0; anonymous speaker id, invented voice name" + }, + "reference": { + "source_filename": "freja.wav", + "published_in_model_repo": false, + "public_preview": "audio/refs/freja.opus", + "sha256": "078baf737c9acc6f567dd87a8b3eca897becb0e83d9209279bf8314830d2678a", + "duration_s": 8.78, + "construction": "single continuous clip, close mic" + }, + "profile": { + "hf_path": "voices/freja.safetensors", + "sha256": "0a3f913b9b47c529393f843850228bd8eee6a883b4c138978fd3afa7758561e8" + }, + "speaker_similarity": 0.907, + "sample": { + "seed": 7, + "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.", + "work": "Den grimme aelling (H. C. Andersen) — public domain", + "audio": "audio/freja.opus", + "sha256": "cd785cc844250a5ae97010e2a3b0ee9260c5a24d016350d931aa7ec776958812" + } + } +]