diff --git a/README.md b/README.md
index 2e87cb0fd10bc83a648f66d53c5f098eb876e350..4db28afa9528f68415538245009b70427e84e4f0 100644
--- a/README.md
+++ b/README.md
@@ -1,15 +1,77 @@
---
-title: Loudkit
-emoji: 📚
-colorFrom: pink
-colorTo: green
+title: loudkit
+emoji: 🔊
+colorFrom: gray
+colorTo: red
sdk: gradio
-sdk_version: 6.26.0
-python_version: '3.12'
+sdk_version: 5.50.0
+python_version: "3.12.12"
app_file: app.py
pinned: false
license: apache-2.0
-short_description: showcase of the loudkit TTS library
+short_description: On-device TTS. Twenty voices, ten languages, one engine.
+models:
+ - loudreader/loudr-1
+preload_from_hub:
+ - loudreader/loudr-1 loudr-1.safetensors,loudr-1-enrollment.safetensors,ve.safetensors,manifest.json,release.json,SHA256SUMS,voices/carmen.safetensors,voices/colette.safetensors,voices/dante.safetensors,voices/darkman.safetensors,voices/dave.safetensors,voices/freja.safetensors,voices/gosia.safetensors,voices/henri.safetensors,voices/ines.safetensors,voices/joe.safetensors,voices/kathleen.safetensors,voices/kerstin.safetensors,voices/nathalie.safetensors,voices/nils.safetensors,voices/paola.safetensors,voices/pim.safetensors,voices/selma.safetensors,voices/soren.safetensors,voices/thorsten.safetensors,voices/tugao.safetensors
---
-Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
+# loudkit
+
+Twenty voices across ten languages, from [loudreader/loudr-1](https://huggingface.co/loudreader/loudr-1),
+running the [loudkit](https://github.com/loudreader/loudkit) engine on ZeroGPU.
+
+## Three tabs
+
+- **Listen.** Twenty voices, each beside the reference recording it was enrolled
+ from. These files were rendered ahead of time and ship in this repo. This tab
+ uses no GPU.
+- **Speak.** Your text, in any of the twenty voices. Up to 1,000 characters.
+- **Clone.** Your own voice, from about ten seconds of audio.
+
+ZeroGPU bills GPU time to the visitor, not to the owner. An anonymous visitor
+gets about two minutes a day. A signed-in free account gets about five. Listening
+costs none of it.
+
+## Cloning and consent
+
+Clone your own voice, or a voice you have permission to use.
+
+- The microphone is the default path.
+- An upload is secondary, and needs an explicit confirmation.
+- Recordings are deleted when the request ends. Nothing is kept.
+
+See [RESPONSIBLE_USE](https://github.com/loudreader/loudkit/blob/main/RESPONSIBLE_USE.md).
+
+## Determinism
+
+The Speak tab has a determinism check. It renders the same text twice at the same
+seed and prints the SHA-256 of both waveforms. They match.
+
+That holds within this build and this device. loudkit promises a bit-identical
+waveform for the same seed, build, backend and input. It does not promise that
+your machine matches this GPU. See the
+[identity contract](https://github.com/loudreader/loudkit/blob/main/docs/reference/IDENTITY-CONTRACT.md).
+
+## Run it locally
+
+```bash
+pip install "loudkit[torch,audio,enroll,hub]"
+```
+
+```python
+import loudkit as lk
+
+engine = lk.load("loudreader/loudr-1")
+voice = lk.voice("kathleen", repo="loudreader/loudr-1")
+engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav")
+```
+
+Audio in this Space, and from `Result.save`, carries C2PA provenance: the
+algorithm fingerprint, the recipe and the seed.
+
+## Voice sources
+
+Every voice is enrolled from a public-domain or openly licensed recording.
+`voices.json` in this repo carries the full record for each one: donor, source,
+licence, consent, and the SHA-256 of both the reference and the sample.
diff --git a/app.py b/app.py
new file mode 100644
index 0000000000000000000000000000000000000000..3e2e44e07163d6b6d66f366ce80678012a99f0f9
--- /dev/null
+++ b/app.py
@@ -0,0 +1,541 @@
+"""The loudkit demo Space: hear twenty voices, speak your own text, clone your own.
+
+ZeroGPU bills GPU time to the *visitor*, not to the owner: an anonymous visitor
+gets about two minutes a day, a signed-in free account about five. A demo whose
+first click spends that budget is one most people bounce off before they have
+heard anything at all. So the Listen tab is twenty pre-rendered files served
+straight out of this repo — no GPU, no queue, no quota — and the GPU is spent
+only on what a visitor types or records.
+
+The engine and the enroller are both built on `cuda` at module level, which is
+what ZeroGPU asks for: CUDA transfers are optimised for start-up placement, and
+lazy-loading inside a `@spaces.GPU` function is explicitly discouraged. Each
+decorated call then runs in a freshly forked, short-lived process, which is also
+why there is no `torch.compile` and no CUDA graph capture here: both pay their
+cost once per process and would never amortise.
+
+Cloning is exposed, which the CPU scaffold this replaces deliberately did not do.
+The reasoning that kept it out was about consent, not about capability, so the
+consent is built into the shape of the tab rather than written beside it: the
+microphone is the default path, an upload is secondary and gated on an explicit
+confirmation, and neither recording outlives the request that carried it.
+"""
+
+from __future__ import annotations
+
+import contextlib
+import dataclasses
+import hashlib
+import json
+import os
+import tempfile
+from pathlib import Path
+
+# Before torch, and before anything that imports torch. The module installs the
+# CUDA emulation that lets a module-level `.to("cuda")` succeed on a machine
+# that has no GPU attached yet.
+import spaces
+
+import gradio as gr
+import numpy as np
+
+import loudkit as lk
+from loudkit.backends.torch_backend import build_torch_enroller
+from loudkit.hub import resolve_enrollment_checkpoint, resolve_voice_encoder
+
+REPO = "loudreader/loudr-1"
+DEVICE = "cuda"
+HERE = Path(__file__).parent
+
+# The CPU scaffold capped text at 300 characters because CPU synthesis ran at
+# roughly a tenth of real time. On a GPU the cap is about the visitor's daily
+# quota instead, which is a far looser bound: 1000 characters is ~70 s of speech.
+MAX_CHARS = 1_000
+MAX_CLONE_CHARS = 400
+MAX_PROBE_CHARS = 200
+
+# The enroller refuses anything over 30 s and wants 5 to 10. The prompt is built
+# from the first 10 s; the speaker embedding reads whatever else is there, so a
+# little past the prompt window is useful and 20 s stays clear of the refusal.
+ENROLL_SECONDS = 20.0
+
+DOCS = "https://github.com/loudreader/loudkit"
+IDENTITY_CONTRACT = f"{DOCS}/blob/main/docs/reference/IDENTITY-CONTRACT.md"
+RESPONSIBLE_USE = f"{DOCS}/blob/main/RESPONSIBLE_USE.md"
+
+ROSTER = json.loads((HERE / "voices.json").read_text(encoding="utf-8"))
+BY_NAME = {entry["name"]: entry for entry in ROSTER}
+ORDERED = sorted(ROSTER, key=lambda e: (e["language"], e["name"]))
+VOICE_CHOICES = [(f"{e['name']} · {e['language']} ({e['gender']})", e["name"]) for e in ORDERED]
+
+# --------------------------------------------------------------------------
+# Module-level model placement, per the ZeroGPU contract.
+# --------------------------------------------------------------------------
+
+engine = lk.load(REPO, device=DEVICE)
+
+# Voice profiles are numpy, not torch, so they are device-agnostic and cost a
+# few hundred kilobytes each. Loading all twenty up front means switching voice
+# in the Speak tab never blocks on a download.
+PROFILES = {entry["name"]: lk.voice(entry["name"], repo=REPO) for entry in ROSTER}
+
+# Enrollment reads the other half of the release: the speech tokenizer and the
+# speaker encoder, which synthesis never touches, plus the utterance voice
+# encoder that sits beside both. `lk.enroll()` builds this per call by design;
+# a Space would pay the load on every clone, so it is built once here instead.
+enroller = build_torch_enroller(
+ str(resolve_enrollment_checkpoint(REPO)),
+ device=DEVICE,
+ voice_encoder_weights=str(resolve_voice_encoder(REPO)),
+)
+
+FINGERPRINT = engine.algorithm.fingerprint()
+
+_LANGUAGE_NAMES = {e["language_id"]: e["language"] for e in ROSTER}
+LANGUAGE_CHOICES = [("Follow the voice", "")] + [
+ (f"{_LANGUAGE_NAMES.get(code, code)} ({code})", code) for code in lk.languages()
+]
+
+# --------------------------------------------------------------------------
+# Helpers
+# --------------------------------------------------------------------------
+
+
+def _sha256_audio(audio: np.ndarray) -> str:
+ """Hash the waveform, not the file.
+
+ `Result.save` appends a C2PA manifest carrying a wall-clock creation time,
+ which the library itself calls the one byte range in which two identical
+ renders may legitimately differ. Hashing the saved WAV would therefore print
+ two different digests for two identical renders and read as a determinism
+ failure. The waveform is what the identity contract makes its promise about,
+ so the waveform is what gets hashed.
+ """
+ return hashlib.sha256(np.ascontiguousarray(audio, dtype=np.float32).tobytes()).hexdigest()
+
+
+def _write(result: lk.Result, *, voice: str, language: str) -> str:
+ out = tempfile.NamedTemporaryFile(suffix=".wav", delete=False)
+ out.close()
+ # Provenance on: the manifest carries the fingerprint, the recipe and the
+ # seed, which is the machine-readable marking a synthetic-speech demo should
+ # be handing out by default.
+ result.save(out.name, voice=voice, language=language)
+ return out.name
+
+
+def _stats(result: lk.Result) -> str:
+ seconds = len(result.audio) / result.sample_rate
+ return (
+ f"**{seconds:.1f} s of audio.** {result.timings.describe(seconds)}\n\n"
+ f"`algo[{result.algorithm_fingerprint}]` · seed `{result.seed}` · "
+ f"speed `{result.speed:g}x` · {result.sample_rate} Hz"
+ )
+
+
+def _estimate(text: str, *, passes: int = 1, overhead: float = 15.0) -> int:
+ """Seconds of GPU to ask for.
+
+ Speech runs at roughly 14 characters a second, and the render is asked to
+ keep up with better than real time; the overhead covers the process fork and
+ the first real CUDA touch. Asking for too much costs queue priority but not
+ quota, which is charged on effective duration, so this leans generous.
+ """
+ audio_seconds = len((text or "").strip()) / 14.0
+ return int(min(180.0, overhead + passes * max(4.0, audio_seconds * 0.9)))
+
+
+def _check(text: str, limit: int) -> str:
+ text = (text or "").strip()
+ if not text:
+ raise gr.Error("Type something to say.")
+ if len(text) > limit:
+ raise gr.Error(f"Keep it under {limit:,} characters here. The library itself takes 10,000.")
+ return text
+
+
+# --------------------------------------------------------------------------
+# Listen. No GPU: these files were rendered ahead of time and ship in the repo.
+# --------------------------------------------------------------------------
+
+
+def listen(name: str):
+ entry = BY_NAME[name]
+ sample, reference, source = entry["sample"], entry["reference"], entry["source"]
+
+ lines = [
+ f"### {entry['name']}. {entry['language']} ({entry['gender']}).",
+ "",
+ f"> {sample['text']}",
+ "",
+ f"From *{sample['work']}*, seed `{sample['seed']}`.",
+ "",
+ f"- Reference recording: {reference['duration_s']:.1f} s, {reference['construction']}.",
+ f"- Source: [{source['name']}]({source['url']}), {source['license']}.",
+ f"- Consent: {source['consent']}.",
+ ]
+ similarity = entry.get("speaker_similarity")
+ if similarity is not None:
+ lines.append(f"- Speaker similarity to the reference: {similarity:.3f}.")
+ lines.append(f"- Voice profile: `{entry['profile']['hf_path']}`.")
+
+ return (
+ str(HERE / sample["audio"]),
+ str(HERE / reference["public_preview"]),
+ "\n".join(lines),
+ )
+
+
+ROSTER_TABLE = [
+ [
+ entry["name"],
+ entry["language"],
+ entry["gender"],
+ entry["source"]["license"],
+ f"{entry['speaker_similarity']:.3f}" if entry.get("speaker_similarity") is not None else "",
+ ]
+ for entry in ORDERED
+]
+
+
+# --------------------------------------------------------------------------
+# Speak. GPU.
+# --------------------------------------------------------------------------
+
+
+def _speak_duration(text, name, language, seed, speed):
+ return _estimate(text, overhead=15.0)
+
+
+@spaces.GPU(duration=_speak_duration)
+def speak(text: str, name: str, language: str, seed: float, speed: float):
+ text = _check(text, MAX_CHARS)
+ result = engine.synthesize_long(
+ text,
+ PROFILES[name],
+ seed=int(seed),
+ language=language or None,
+ speed=float(speed),
+ )
+ label = language or BY_NAME[name]["language_id"]
+ return _write(result, voice=name, language=label), _stats(result)
+
+
+# --------------------------------------------------------------------------
+# Clone. GPU. The microphone is the default path; an upload is gated.
+# --------------------------------------------------------------------------
+
+
+def _clone_duration(mic, upload, consent, text, language, seed, speed):
+ # Enrollment is a fixed cost on top of the render: two encoders and a
+ # tokenizer over at most 20 s of audio.
+ return _estimate(text, overhead=30.0)
+
+
+@spaces.GPU(duration=_clone_duration)
+def clone(mic, upload, consent: bool, text: str, language: str, seed: float, speed: float):
+ source = mic or upload
+ if not source:
+ raise gr.Error("Record yourself first, or upload a clip you are allowed to use.")
+ if upload and not mic and not consent:
+ raise gr.Error("Confirm the uploaded voice is yours, or that you have permission to use it.")
+ text = _check(text, MAX_CLONE_CHARS)
+
+ try:
+ import librosa
+
+ samples, _ = librosa.load(source, sr=24_000, mono=True)
+ limit = int(ENROLL_SECONDS * 24_000)
+ if samples.size > limit:
+ samples = samples[:limit]
+
+ try:
+ profile = enroller.enroll(samples, 24_000, name="your voice")
+ except ValueError as exc:
+ # The library's own messages name the bound and describe a good
+ # input, which is more useful than anything restated here.
+ raise gr.Error(str(exc)) from exc
+
+ # `enroll` writes no language, so every cloned voice would claim English
+ # and read its text through the English funnel.
+ profile = dataclasses.replace(profile, language=language or "en")
+
+ result = engine.synthesize_long(
+ text, profile, seed=int(seed), language=language or None, speed=float(speed)
+ )
+ return _write(result, voice="cloned", language=profile.language), _stats(result)
+ finally:
+ # Nothing the visitor recorded outlives the request that carried it.
+ with contextlib.suppress(OSError):
+ os.unlink(source)
+
+
+# --------------------------------------------------------------------------
+# Determinism probe. GPU. Renders the same text twice at the same seed.
+# --------------------------------------------------------------------------
+
+
+def _probe_duration(text, name, seed):
+ return _estimate(text, passes=2, overhead=20.0)
+
+
+@spaces.GPU(duration=_probe_duration)
+def probe(text: str, name: str, seed: float):
+ text = _check(text, MAX_PROBE_CHARS)
+ profile = PROFILES[name]
+ first = engine.synthesize_long(text, profile, seed=int(seed))
+ second = engine.synthesize_long(text, profile, seed=int(seed))
+
+ left, right = _sha256_audio(first.audio), _sha256_audio(second.audio)
+ verdict = "Identical." if left == right else "Different. Please report this."
+
+ return "\n".join(
+ [
+ f"**{verdict}**",
+ "",
+ "```",
+ f"render 1 sha256 {left}",
+ f"render 2 sha256 {right}",
+ f" algo[{first.algorithm_fingerprint}] seed {int(seed)}",
+ "```",
+ "",
+ "Identical within this build and this device. loudkit promises a "
+ "bit-identical waveform for the same seed, build, backend and input. "
+ "It does not promise that your laptop matches this GPU. "
+ f"[Read the identity contract]({IDENTITY_CONTRACT}).",
+ ]
+ )
+
+
+# --------------------------------------------------------------------------
+# Interface
+# --------------------------------------------------------------------------
+
+# loudreader.io: cream ground, ink text, black pill buttons at 14px.
+CSS = """
+#lk-head h1 { font-size: 2.15rem; margin-bottom: .25rem; letter-spacing: -.02em; }
+#lk-head p { margin-top: 0; }
+.lk-pill {
+ display: inline-block; padding: .2rem .75rem; margin: .15rem .35rem .15rem 0;
+ border: 1px solid #ded8ce; border-radius: 999px; font-size: .8rem;
+ color: #374151; background: #fffdfa;
+}
+.lk-card { background: #fffdfa; border: 1px solid #e7e1d7; border-radius: 14px; padding: .35rem 1rem; }
+footer { display: none !important; }
+"""
+
+# Gradio follows the visitor's system theme unless told otherwise, and this
+# palette is light-first. Without this the ink-on-cream tokens below land under
+# a dark stylesheet and the text turns near-white on a cream ground.
+FORCE_LIGHT = """
+() => {
+ const url = new URL(window.location);
+ if (url.searchParams.get('__theme') !== 'light') {
+ url.searchParams.set('__theme', 'light');
+ window.location.replace(url.href);
+ }
+}
+"""
+
+THEME = gr.themes.Soft(
+ primary_hue=gr.themes.colors.gray,
+ neutral_hue=gr.themes.colors.stone,
+ font=[gr.themes.GoogleFont("Inter"), "system-ui", "sans-serif"],
+).set(
+ body_background_fill="#f7f5f2",
+ body_text_color="#111827",
+ body_text_color_subdued="#4b5563",
+ block_background_fill="#fffdfa",
+ block_border_color="#e7e1d7",
+ border_color_primary="#e7e1d7",
+ input_background_fill="#ffffff",
+ button_primary_background_fill="#111827",
+ button_primary_background_fill_hover="#374151",
+ button_primary_text_color="#ffffff",
+ button_large_radius="14px",
+ button_small_radius="14px",
+)
+
+with gr.Blocks(title="loudkit", theme=THEME, css=CSS, js=FORCE_LIGHT, fill_width=False) as demo:
+ gr.Markdown(
+ f"""
+# Twenty voices. Ten languages. One engine.
+
+On-device text to speech, running here on ZeroGPU.
+[Model](https://huggingface.co/{REPO}) · [Code]({DOCS}) · [Responsible use]({RESPONSIBLE_USE})
+
+Listening costs no GPU
+Speaking and cloning spend your daily quota
+algo[{FINGERPRINT}]
+""",
+ elem_id="lk-head",
+ )
+
+ with gr.Tabs():
+ # ---------------- Listen ----------------
+ with gr.Tab("Listen"):
+ gr.Markdown(
+ "Twenty voices, rendered ahead of time and served as files. "
+ "This tab uses no GPU and spends none of your quota. "
+ "Each voice is paired with the reference recording it was enrolled from."
+ )
+ with gr.Row():
+ with gr.Column(scale=1):
+ pick = gr.Dropdown(
+ VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True
+ )
+ made = gr.Audio(label="loudkit", type="filepath", interactive=False)
+ ref = gr.Audio(label="Reference recording", type="filepath", interactive=False)
+ with gr.Column(scale=1):
+ card = gr.Markdown(elem_classes="lk-card")
+
+ with gr.Accordion("The whole roster", open=False):
+ gr.Dataframe(
+ value=ROSTER_TABLE,
+ headers=["Voice", "Language", "Gender", "Licence", "Similarity"],
+ interactive=False,
+ wrap=True,
+ )
+
+ pick.change(listen, pick, [made, ref, card])
+ demo.load(listen, pick, [made, ref, card])
+
+ # ---------------- Speak ----------------
+ with gr.Tab("Speak"):
+ gr.Markdown(
+ f"Your text, in one of the twenty voices. "
+ f"Up to {MAX_CHARS:,} characters here. The library itself takes 10,000. "
+ "This tab spends your ZeroGPU quota."
+ )
+ with gr.Row():
+ with gr.Column(scale=3):
+ say = gr.Textbox(
+ label="Text",
+ placeholder="Hello from loudkit.",
+ lines=4,
+ max_length=MAX_CHARS,
+ )
+ with gr.Column(scale=2):
+ say_voice = gr.Dropdown(
+ VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True
+ )
+ say_lang = gr.Dropdown(
+ LANGUAGE_CHOICES, value="", label="Read the text as"
+ )
+ with gr.Row():
+ say_seed = gr.Number(value=7, precision=0, label="Seed")
+ say_speed = gr.Slider(
+ lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed"
+ )
+ say_go = gr.Button("Speak", variant="primary")
+ say_out = gr.Audio(label="Speech", type="filepath")
+ say_stats = gr.Markdown()
+
+ say_go.click(
+ speak,
+ [say, say_voice, say_lang, say_seed, say_speed],
+ [say_out, say_stats],
+ )
+
+ with gr.Accordion("Determinism check", open=False):
+ gr.Markdown(
+ "This renders the same text twice at the same seed and hashes "
+ "both waveforms. The digests must match."
+ )
+ with gr.Row():
+ probe_text = gr.Textbox(
+ value="The same seed gives the same audio.",
+ label="Text",
+ max_length=MAX_PROBE_CHARS,
+ scale=3,
+ )
+ probe_voice = gr.Dropdown(
+ VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", scale=2
+ )
+ probe_seed = gr.Number(value=7, precision=0, label="Seed", scale=1)
+ probe_go = gr.Button("Render twice")
+ probe_out = gr.Markdown()
+ probe_go.click(probe, [probe_text, probe_voice, probe_seed], probe_out)
+
+ # ---------------- Clone ----------------
+ with gr.Tab("Clone"):
+ gr.Markdown(
+ f"""
+Clone a voice from a short recording, then speak with it.
+
+- Record 5 to 10 seconds. Read anything. Speak normally.
+- Clone only your own voice, or a voice you have permission to use.
+- Nothing you record is kept. The recording is deleted when the request ends.
+- See [Responsible use]({RESPONSIBLE_USE}).
+"""
+ )
+ with gr.Row():
+ with gr.Column(scale=1):
+ mic = gr.Audio(
+ sources=["microphone"],
+ type="filepath",
+ label="Record yourself",
+ )
+ with gr.Accordion("Upload a file instead", open=False):
+ upload = gr.Audio(
+ sources=["upload"], type="filepath", label="Audio file"
+ )
+ consent = gr.Checkbox(
+ value=False,
+ label=(
+ "This is my own voice, or I have permission from the "
+ "person who owns it."
+ ),
+ )
+ with gr.Column(scale=1):
+ clone_text = gr.Textbox(
+ label="Text to speak",
+ placeholder="Now in my own voice.",
+ lines=3,
+ max_length=MAX_CLONE_CHARS,
+ )
+ clone_lang = gr.Dropdown(
+ LANGUAGE_CHOICES[1:], value="en", label="Language of the text"
+ )
+ with gr.Row():
+ clone_seed = gr.Number(value=7, precision=0, label="Seed")
+ clone_speed = gr.Slider(
+ lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed"
+ )
+ clone_go = gr.Button("Clone and speak", variant="primary")
+ clone_out = gr.Audio(label="Speech", type="filepath")
+ clone_stats = gr.Markdown()
+
+ clone_go.click(
+ clone,
+ [mic, upload, consent, clone_text, clone_lang, clone_seed, clone_speed],
+ [clone_out, clone_stats],
+ )
+
+ gr.Markdown(
+ f"""
+---
+Run the same engine locally, where nothing is queued and nothing is metered.
+
+```bash
+pip install "loudkit[torch,audio,enroll,hub]"
+```
+
+```python
+import loudkit as lk
+
+engine = lk.load("{REPO}")
+voice = lk.voice("kathleen", repo="{REPO}")
+engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav")
+```
+
+Output files carry C2PA provenance: the fingerprint, the recipe and the seed.
+"""
+ )
+
+# The engine holds one set of weights and renders with an internal producer
+# thread. One render at a time keeps two requests off the same buffers.
+demo.queue(default_concurrency_limit=1, max_size=24)
+
+if __name__ == "__main__":
+ demo.launch()
diff --git a/audio/carmen.opus b/audio/carmen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..e1e18609163ff3cd67315d85497818f104d33a58
Binary files /dev/null and b/audio/carmen.opus differ
diff --git a/audio/colette.opus b/audio/colette.opus
new file mode 100644
index 0000000000000000000000000000000000000000..682389d876f822ca755d906f05db376a18ff68d9
Binary files /dev/null and b/audio/colette.opus differ
diff --git a/audio/dante.opus b/audio/dante.opus
new file mode 100644
index 0000000000000000000000000000000000000000..6628f6858d4f71d379c47411baee9507f8192928
Binary files /dev/null and b/audio/dante.opus differ
diff --git a/audio/darkman.opus b/audio/darkman.opus
new file mode 100644
index 0000000000000000000000000000000000000000..7a6ce2e7c75e4206475b532df2fe8a9850a29929
Binary files /dev/null and b/audio/darkman.opus differ
diff --git a/audio/dave.opus b/audio/dave.opus
new file mode 100644
index 0000000000000000000000000000000000000000..502d2da2e6fdf140c6310a3336457eb210a7bc31
Binary files /dev/null and b/audio/dave.opus differ
diff --git a/audio/freja.opus b/audio/freja.opus
new file mode 100644
index 0000000000000000000000000000000000000000..2160a36de61672adf91284b9c74c692b59f07206
Binary files /dev/null and b/audio/freja.opus differ
diff --git a/audio/gosia.opus b/audio/gosia.opus
new file mode 100644
index 0000000000000000000000000000000000000000..853a342df5107a4894770834ef16a8539ecf1f6f
Binary files /dev/null and b/audio/gosia.opus differ
diff --git a/audio/henri.opus b/audio/henri.opus
new file mode 100644
index 0000000000000000000000000000000000000000..97c0f1a5a9bd18e1516d97a70f26fe3a5b3ae0bb
Binary files /dev/null and b/audio/henri.opus differ
diff --git a/audio/ines.opus b/audio/ines.opus
new file mode 100644
index 0000000000000000000000000000000000000000..2f988d7adb9dc0bd7041089ac470d0cbee02c832
Binary files /dev/null and b/audio/ines.opus differ
diff --git a/audio/joe.opus b/audio/joe.opus
new file mode 100644
index 0000000000000000000000000000000000000000..d35624d4b0cd92eb44d278737e92a65c95db9892
Binary files /dev/null and b/audio/joe.opus differ
diff --git a/audio/kathleen.opus b/audio/kathleen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..afab998420c69de122c666484f135471301633ef
Binary files /dev/null and b/audio/kathleen.opus differ
diff --git a/audio/kerstin.opus b/audio/kerstin.opus
new file mode 100644
index 0000000000000000000000000000000000000000..bc9b99ead23656b1e5f2ca7af21146bc1923b6ef
Binary files /dev/null and b/audio/kerstin.opus differ
diff --git a/audio/nathalie.opus b/audio/nathalie.opus
new file mode 100644
index 0000000000000000000000000000000000000000..78d74b015d59796e31d798558ffe6c846d4e5619
Binary files /dev/null and b/audio/nathalie.opus differ
diff --git a/audio/nils.opus b/audio/nils.opus
new file mode 100644
index 0000000000000000000000000000000000000000..9f317ca8b1aba3b9aa3dd0e86719cda67125b6d8
Binary files /dev/null and b/audio/nils.opus differ
diff --git a/audio/paola.opus b/audio/paola.opus
new file mode 100644
index 0000000000000000000000000000000000000000..6f936ee395156750e75bbb580eb239a62b32eb3a
Binary files /dev/null and b/audio/paola.opus differ
diff --git a/audio/pim.opus b/audio/pim.opus
new file mode 100644
index 0000000000000000000000000000000000000000..5d31d74acd05d2cc5a8b191d3c78f06387e1eb05
Binary files /dev/null and b/audio/pim.opus differ
diff --git a/audio/refs/carmen.opus b/audio/refs/carmen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..220d1a403a3b535d284ce5c4008e3a5fcb736945
Binary files /dev/null and b/audio/refs/carmen.opus differ
diff --git a/audio/refs/colette.opus b/audio/refs/colette.opus
new file mode 100644
index 0000000000000000000000000000000000000000..dce18d4f09e5b07e10f9c46441fca118b4d17b67
Binary files /dev/null and b/audio/refs/colette.opus differ
diff --git a/audio/refs/dante.opus b/audio/refs/dante.opus
new file mode 100644
index 0000000000000000000000000000000000000000..b58368afe0db03b2d2f960c75d91280146d5f901
Binary files /dev/null and b/audio/refs/dante.opus differ
diff --git a/audio/refs/darkman.opus b/audio/refs/darkman.opus
new file mode 100644
index 0000000000000000000000000000000000000000..cb98a474936dbbdb79b3020552e955242f61ef0d
Binary files /dev/null and b/audio/refs/darkman.opus differ
diff --git a/audio/refs/dave.opus b/audio/refs/dave.opus
new file mode 100644
index 0000000000000000000000000000000000000000..ca3c66e199e7bd900d1d4180aa91b65aaba5d9e8
Binary files /dev/null and b/audio/refs/dave.opus differ
diff --git a/audio/refs/freja.opus b/audio/refs/freja.opus
new file mode 100644
index 0000000000000000000000000000000000000000..adbb18927dc20821a83eb11190e9faa2aa50efb3
Binary files /dev/null and b/audio/refs/freja.opus differ
diff --git a/audio/refs/gosia.opus b/audio/refs/gosia.opus
new file mode 100644
index 0000000000000000000000000000000000000000..ac8e283877021b29edb5bd8e27c296973cae3c07
Binary files /dev/null and b/audio/refs/gosia.opus differ
diff --git a/audio/refs/henri.opus b/audio/refs/henri.opus
new file mode 100644
index 0000000000000000000000000000000000000000..59d6873d132f1a4b61081a27cb1817fa1d463bbe
Binary files /dev/null and b/audio/refs/henri.opus differ
diff --git a/audio/refs/ines.opus b/audio/refs/ines.opus
new file mode 100644
index 0000000000000000000000000000000000000000..b31012129731dabfcdd0829dabb54a8e09d5914f
Binary files /dev/null and b/audio/refs/ines.opus differ
diff --git a/audio/refs/joe.opus b/audio/refs/joe.opus
new file mode 100644
index 0000000000000000000000000000000000000000..732465cbc480985e84fe73887ca79a520a33e374
Binary files /dev/null and b/audio/refs/joe.opus differ
diff --git a/audio/refs/kathleen.opus b/audio/refs/kathleen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..03fa0d9136a7614ef66b4d9627042992e6904be2
Binary files /dev/null and b/audio/refs/kathleen.opus differ
diff --git a/audio/refs/kerstin.opus b/audio/refs/kerstin.opus
new file mode 100644
index 0000000000000000000000000000000000000000..d6ca625df4b5c103120b206c35af9cc2c7b47936
Binary files /dev/null and b/audio/refs/kerstin.opus differ
diff --git a/audio/refs/nathalie.opus b/audio/refs/nathalie.opus
new file mode 100644
index 0000000000000000000000000000000000000000..5bc6b148d91e82fa1caa46ce33993fd078319262
Binary files /dev/null and b/audio/refs/nathalie.opus differ
diff --git a/audio/refs/nils.opus b/audio/refs/nils.opus
new file mode 100644
index 0000000000000000000000000000000000000000..bdfc6564eeeb9a75cf8f03f202369a2ae0604380
Binary files /dev/null and b/audio/refs/nils.opus differ
diff --git a/audio/refs/paola.opus b/audio/refs/paola.opus
new file mode 100644
index 0000000000000000000000000000000000000000..cef273a388c73c7e2e3f370ce4af34b56c954344
Binary files /dev/null and b/audio/refs/paola.opus differ
diff --git a/audio/refs/pim.opus b/audio/refs/pim.opus
new file mode 100644
index 0000000000000000000000000000000000000000..df2d2719e8e384f380a958142f3d700406ce6e61
Binary files /dev/null and b/audio/refs/pim.opus differ
diff --git a/audio/refs/selma.opus b/audio/refs/selma.opus
new file mode 100644
index 0000000000000000000000000000000000000000..d6bfaab9d0c0da5e7fe91eb4d2e799e388c21c89
Binary files /dev/null and b/audio/refs/selma.opus differ
diff --git a/audio/refs/soren.opus b/audio/refs/soren.opus
new file mode 100644
index 0000000000000000000000000000000000000000..dda8de236b964dbe5fd8b28e1311cb7fe65540d8
Binary files /dev/null and b/audio/refs/soren.opus differ
diff --git a/audio/refs/thorsten.opus b/audio/refs/thorsten.opus
new file mode 100644
index 0000000000000000000000000000000000000000..5e89b94fbd091c56363285eeebc1fe81deedccac
Binary files /dev/null and b/audio/refs/thorsten.opus differ
diff --git a/audio/refs/tugao.opus b/audio/refs/tugao.opus
new file mode 100644
index 0000000000000000000000000000000000000000..3df5f5af70b92a74f4a006e1a75581f82d027328
Binary files /dev/null and b/audio/refs/tugao.opus differ
diff --git a/audio/selma.opus b/audio/selma.opus
new file mode 100644
index 0000000000000000000000000000000000000000..635b80f10f7832625903ac39d8bf0e831df43eb8
Binary files /dev/null and b/audio/selma.opus differ
diff --git a/audio/soren.opus b/audio/soren.opus
new file mode 100644
index 0000000000000000000000000000000000000000..2ed33a17736947e4ed72c1b4c5fc05b93cc548a9
Binary files /dev/null and b/audio/soren.opus differ
diff --git a/audio/thorsten.opus b/audio/thorsten.opus
new file mode 100644
index 0000000000000000000000000000000000000000..278cbffa5785f98db6cd5fa1394f1eff47e6f5ef
Binary files /dev/null and b/audio/thorsten.opus differ
diff --git a/audio/tugao.opus b/audio/tugao.opus
new file mode 100644
index 0000000000000000000000000000000000000000..71f242f67742e53eecfaeb6363a4779696d3d48c
Binary files /dev/null and b/audio/tugao.opus differ
diff --git a/hf-loudkit/.gitattributes b/hf-loudkit/.gitattributes
new file mode 100644
index 0000000000000000000000000000000000000000..a6344aac8c09253b3b630fb776ae94478aa0275b
--- /dev/null
+++ b/hf-loudkit/.gitattributes
@@ -0,0 +1,35 @@
+*.7z filter=lfs diff=lfs merge=lfs -text
+*.arrow filter=lfs diff=lfs merge=lfs -text
+*.bin filter=lfs diff=lfs merge=lfs -text
+*.bz2 filter=lfs diff=lfs merge=lfs -text
+*.ckpt filter=lfs diff=lfs merge=lfs -text
+*.ftz filter=lfs diff=lfs merge=lfs -text
+*.gz filter=lfs diff=lfs merge=lfs -text
+*.h5 filter=lfs diff=lfs merge=lfs -text
+*.joblib filter=lfs diff=lfs merge=lfs -text
+*.lfs.* filter=lfs diff=lfs merge=lfs -text
+*.mlmodel filter=lfs diff=lfs merge=lfs -text
+*.model filter=lfs diff=lfs merge=lfs -text
+*.msgpack filter=lfs diff=lfs merge=lfs -text
+*.npy filter=lfs diff=lfs merge=lfs -text
+*.npz filter=lfs diff=lfs merge=lfs -text
+*.onnx filter=lfs diff=lfs merge=lfs -text
+*.ot filter=lfs diff=lfs merge=lfs -text
+*.parquet filter=lfs diff=lfs merge=lfs -text
+*.pb filter=lfs diff=lfs merge=lfs -text
+*.pickle filter=lfs diff=lfs merge=lfs -text
+*.pkl filter=lfs diff=lfs merge=lfs -text
+*.pt filter=lfs diff=lfs merge=lfs -text
+*.pth filter=lfs diff=lfs merge=lfs -text
+*.rar filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+saved_model/**/* filter=lfs diff=lfs merge=lfs -text
+*.tar.* filter=lfs diff=lfs merge=lfs -text
+*.tar filter=lfs diff=lfs merge=lfs -text
+*.tflite filter=lfs diff=lfs merge=lfs -text
+*.tgz filter=lfs diff=lfs merge=lfs -text
+*.wasm filter=lfs diff=lfs merge=lfs -text
+*.xz filter=lfs diff=lfs merge=lfs -text
+*.zip filter=lfs diff=lfs merge=lfs -text
+*.zst filter=lfs diff=lfs merge=lfs -text
+*tfevents* filter=lfs diff=lfs merge=lfs -text
diff --git a/hf-loudkit/README.md b/hf-loudkit/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..4db28afa9528f68415538245009b70427e84e4f0
--- /dev/null
+++ b/hf-loudkit/README.md
@@ -0,0 +1,77 @@
+---
+title: loudkit
+emoji: 🔊
+colorFrom: gray
+colorTo: red
+sdk: gradio
+sdk_version: 5.50.0
+python_version: "3.12.12"
+app_file: app.py
+pinned: false
+license: apache-2.0
+short_description: On-device TTS. Twenty voices, ten languages, one engine.
+models:
+ - loudreader/loudr-1
+preload_from_hub:
+ - loudreader/loudr-1 loudr-1.safetensors,loudr-1-enrollment.safetensors,ve.safetensors,manifest.json,release.json,SHA256SUMS,voices/carmen.safetensors,voices/colette.safetensors,voices/dante.safetensors,voices/darkman.safetensors,voices/dave.safetensors,voices/freja.safetensors,voices/gosia.safetensors,voices/henri.safetensors,voices/ines.safetensors,voices/joe.safetensors,voices/kathleen.safetensors,voices/kerstin.safetensors,voices/nathalie.safetensors,voices/nils.safetensors,voices/paola.safetensors,voices/pim.safetensors,voices/selma.safetensors,voices/soren.safetensors,voices/thorsten.safetensors,voices/tugao.safetensors
+---
+
+# loudkit
+
+Twenty voices across ten languages, from [loudreader/loudr-1](https://huggingface.co/loudreader/loudr-1),
+running the [loudkit](https://github.com/loudreader/loudkit) engine on ZeroGPU.
+
+## Three tabs
+
+- **Listen.** Twenty voices, each beside the reference recording it was enrolled
+ from. These files were rendered ahead of time and ship in this repo. This tab
+ uses no GPU.
+- **Speak.** Your text, in any of the twenty voices. Up to 1,000 characters.
+- **Clone.** Your own voice, from about ten seconds of audio.
+
+ZeroGPU bills GPU time to the visitor, not to the owner. An anonymous visitor
+gets about two minutes a day. A signed-in free account gets about five. Listening
+costs none of it.
+
+## Cloning and consent
+
+Clone your own voice, or a voice you have permission to use.
+
+- The microphone is the default path.
+- An upload is secondary, and needs an explicit confirmation.
+- Recordings are deleted when the request ends. Nothing is kept.
+
+See [RESPONSIBLE_USE](https://github.com/loudreader/loudkit/blob/main/RESPONSIBLE_USE.md).
+
+## Determinism
+
+The Speak tab has a determinism check. It renders the same text twice at the same
+seed and prints the SHA-256 of both waveforms. They match.
+
+That holds within this build and this device. loudkit promises a bit-identical
+waveform for the same seed, build, backend and input. It does not promise that
+your machine matches this GPU. See the
+[identity contract](https://github.com/loudreader/loudkit/blob/main/docs/reference/IDENTITY-CONTRACT.md).
+
+## Run it locally
+
+```bash
+pip install "loudkit[torch,audio,enroll,hub]"
+```
+
+```python
+import loudkit as lk
+
+engine = lk.load("loudreader/loudr-1")
+voice = lk.voice("kathleen", repo="loudreader/loudr-1")
+engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav")
+```
+
+Audio in this Space, and from `Result.save`, carries C2PA provenance: the
+algorithm fingerprint, the recipe and the seed.
+
+## Voice sources
+
+Every voice is enrolled from a public-domain or openly licensed recording.
+`voices.json` in this repo carries the full record for each one: donor, source,
+licence, consent, and the SHA-256 of both the reference and the sample.
diff --git a/hf-loudkit/app.py b/hf-loudkit/app.py
new file mode 100644
index 0000000000000000000000000000000000000000..3e2e44e07163d6b6d66f366ce80678012a99f0f9
--- /dev/null
+++ b/hf-loudkit/app.py
@@ -0,0 +1,541 @@
+"""The loudkit demo Space: hear twenty voices, speak your own text, clone your own.
+
+ZeroGPU bills GPU time to the *visitor*, not to the owner: an anonymous visitor
+gets about two minutes a day, a signed-in free account about five. A demo whose
+first click spends that budget is one most people bounce off before they have
+heard anything at all. So the Listen tab is twenty pre-rendered files served
+straight out of this repo — no GPU, no queue, no quota — and the GPU is spent
+only on what a visitor types or records.
+
+The engine and the enroller are both built on `cuda` at module level, which is
+what ZeroGPU asks for: CUDA transfers are optimised for start-up placement, and
+lazy-loading inside a `@spaces.GPU` function is explicitly discouraged. Each
+decorated call then runs in a freshly forked, short-lived process, which is also
+why there is no `torch.compile` and no CUDA graph capture here: both pay their
+cost once per process and would never amortise.
+
+Cloning is exposed, which the CPU scaffold this replaces deliberately did not do.
+The reasoning that kept it out was about consent, not about capability, so the
+consent is built into the shape of the tab rather than written beside it: the
+microphone is the default path, an upload is secondary and gated on an explicit
+confirmation, and neither recording outlives the request that carried it.
+"""
+
+from __future__ import annotations
+
+import contextlib
+import dataclasses
+import hashlib
+import json
+import os
+import tempfile
+from pathlib import Path
+
+# Before torch, and before anything that imports torch. The module installs the
+# CUDA emulation that lets a module-level `.to("cuda")` succeed on a machine
+# that has no GPU attached yet.
+import spaces
+
+import gradio as gr
+import numpy as np
+
+import loudkit as lk
+from loudkit.backends.torch_backend import build_torch_enroller
+from loudkit.hub import resolve_enrollment_checkpoint, resolve_voice_encoder
+
+REPO = "loudreader/loudr-1"
+DEVICE = "cuda"
+HERE = Path(__file__).parent
+
+# The CPU scaffold capped text at 300 characters because CPU synthesis ran at
+# roughly a tenth of real time. On a GPU the cap is about the visitor's daily
+# quota instead, which is a far looser bound: 1000 characters is ~70 s of speech.
+MAX_CHARS = 1_000
+MAX_CLONE_CHARS = 400
+MAX_PROBE_CHARS = 200
+
+# The enroller refuses anything over 30 s and wants 5 to 10. The prompt is built
+# from the first 10 s; the speaker embedding reads whatever else is there, so a
+# little past the prompt window is useful and 20 s stays clear of the refusal.
+ENROLL_SECONDS = 20.0
+
+DOCS = "https://github.com/loudreader/loudkit"
+IDENTITY_CONTRACT = f"{DOCS}/blob/main/docs/reference/IDENTITY-CONTRACT.md"
+RESPONSIBLE_USE = f"{DOCS}/blob/main/RESPONSIBLE_USE.md"
+
+ROSTER = json.loads((HERE / "voices.json").read_text(encoding="utf-8"))
+BY_NAME = {entry["name"]: entry for entry in ROSTER}
+ORDERED = sorted(ROSTER, key=lambda e: (e["language"], e["name"]))
+VOICE_CHOICES = [(f"{e['name']} · {e['language']} ({e['gender']})", e["name"]) for e in ORDERED]
+
+# --------------------------------------------------------------------------
+# Module-level model placement, per the ZeroGPU contract.
+# --------------------------------------------------------------------------
+
+engine = lk.load(REPO, device=DEVICE)
+
+# Voice profiles are numpy, not torch, so they are device-agnostic and cost a
+# few hundred kilobytes each. Loading all twenty up front means switching voice
+# in the Speak tab never blocks on a download.
+PROFILES = {entry["name"]: lk.voice(entry["name"], repo=REPO) for entry in ROSTER}
+
+# Enrollment reads the other half of the release: the speech tokenizer and the
+# speaker encoder, which synthesis never touches, plus the utterance voice
+# encoder that sits beside both. `lk.enroll()` builds this per call by design;
+# a Space would pay the load on every clone, so it is built once here instead.
+enroller = build_torch_enroller(
+ str(resolve_enrollment_checkpoint(REPO)),
+ device=DEVICE,
+ voice_encoder_weights=str(resolve_voice_encoder(REPO)),
+)
+
+FINGERPRINT = engine.algorithm.fingerprint()
+
+_LANGUAGE_NAMES = {e["language_id"]: e["language"] for e in ROSTER}
+LANGUAGE_CHOICES = [("Follow the voice", "")] + [
+ (f"{_LANGUAGE_NAMES.get(code, code)} ({code})", code) for code in lk.languages()
+]
+
+# --------------------------------------------------------------------------
+# Helpers
+# --------------------------------------------------------------------------
+
+
+def _sha256_audio(audio: np.ndarray) -> str:
+ """Hash the waveform, not the file.
+
+ `Result.save` appends a C2PA manifest carrying a wall-clock creation time,
+ which the library itself calls the one byte range in which two identical
+ renders may legitimately differ. Hashing the saved WAV would therefore print
+ two different digests for two identical renders and read as a determinism
+ failure. The waveform is what the identity contract makes its promise about,
+ so the waveform is what gets hashed.
+ """
+ return hashlib.sha256(np.ascontiguousarray(audio, dtype=np.float32).tobytes()).hexdigest()
+
+
+def _write(result: lk.Result, *, voice: str, language: str) -> str:
+ out = tempfile.NamedTemporaryFile(suffix=".wav", delete=False)
+ out.close()
+ # Provenance on: the manifest carries the fingerprint, the recipe and the
+ # seed, which is the machine-readable marking a synthetic-speech demo should
+ # be handing out by default.
+ result.save(out.name, voice=voice, language=language)
+ return out.name
+
+
+def _stats(result: lk.Result) -> str:
+ seconds = len(result.audio) / result.sample_rate
+ return (
+ f"**{seconds:.1f} s of audio.** {result.timings.describe(seconds)}\n\n"
+ f"`algo[{result.algorithm_fingerprint}]` · seed `{result.seed}` · "
+ f"speed `{result.speed:g}x` · {result.sample_rate} Hz"
+ )
+
+
+def _estimate(text: str, *, passes: int = 1, overhead: float = 15.0) -> int:
+ """Seconds of GPU to ask for.
+
+ Speech runs at roughly 14 characters a second, and the render is asked to
+ keep up with better than real time; the overhead covers the process fork and
+ the first real CUDA touch. Asking for too much costs queue priority but not
+ quota, which is charged on effective duration, so this leans generous.
+ """
+ audio_seconds = len((text or "").strip()) / 14.0
+ return int(min(180.0, overhead + passes * max(4.0, audio_seconds * 0.9)))
+
+
+def _check(text: str, limit: int) -> str:
+ text = (text or "").strip()
+ if not text:
+ raise gr.Error("Type something to say.")
+ if len(text) > limit:
+ raise gr.Error(f"Keep it under {limit:,} characters here. The library itself takes 10,000.")
+ return text
+
+
+# --------------------------------------------------------------------------
+# Listen. No GPU: these files were rendered ahead of time and ship in the repo.
+# --------------------------------------------------------------------------
+
+
+def listen(name: str):
+ entry = BY_NAME[name]
+ sample, reference, source = entry["sample"], entry["reference"], entry["source"]
+
+ lines = [
+ f"### {entry['name']}. {entry['language']} ({entry['gender']}).",
+ "",
+ f"> {sample['text']}",
+ "",
+ f"From *{sample['work']}*, seed `{sample['seed']}`.",
+ "",
+ f"- Reference recording: {reference['duration_s']:.1f} s, {reference['construction']}.",
+ f"- Source: [{source['name']}]({source['url']}), {source['license']}.",
+ f"- Consent: {source['consent']}.",
+ ]
+ similarity = entry.get("speaker_similarity")
+ if similarity is not None:
+ lines.append(f"- Speaker similarity to the reference: {similarity:.3f}.")
+ lines.append(f"- Voice profile: `{entry['profile']['hf_path']}`.")
+
+ return (
+ str(HERE / sample["audio"]),
+ str(HERE / reference["public_preview"]),
+ "\n".join(lines),
+ )
+
+
+ROSTER_TABLE = [
+ [
+ entry["name"],
+ entry["language"],
+ entry["gender"],
+ entry["source"]["license"],
+ f"{entry['speaker_similarity']:.3f}" if entry.get("speaker_similarity") is not None else "",
+ ]
+ for entry in ORDERED
+]
+
+
+# --------------------------------------------------------------------------
+# Speak. GPU.
+# --------------------------------------------------------------------------
+
+
+def _speak_duration(text, name, language, seed, speed):
+ return _estimate(text, overhead=15.0)
+
+
+@spaces.GPU(duration=_speak_duration)
+def speak(text: str, name: str, language: str, seed: float, speed: float):
+ text = _check(text, MAX_CHARS)
+ result = engine.synthesize_long(
+ text,
+ PROFILES[name],
+ seed=int(seed),
+ language=language or None,
+ speed=float(speed),
+ )
+ label = language or BY_NAME[name]["language_id"]
+ return _write(result, voice=name, language=label), _stats(result)
+
+
+# --------------------------------------------------------------------------
+# Clone. GPU. The microphone is the default path; an upload is gated.
+# --------------------------------------------------------------------------
+
+
+def _clone_duration(mic, upload, consent, text, language, seed, speed):
+ # Enrollment is a fixed cost on top of the render: two encoders and a
+ # tokenizer over at most 20 s of audio.
+ return _estimate(text, overhead=30.0)
+
+
+@spaces.GPU(duration=_clone_duration)
+def clone(mic, upload, consent: bool, text: str, language: str, seed: float, speed: float):
+ source = mic or upload
+ if not source:
+ raise gr.Error("Record yourself first, or upload a clip you are allowed to use.")
+ if upload and not mic and not consent:
+ raise gr.Error("Confirm the uploaded voice is yours, or that you have permission to use it.")
+ text = _check(text, MAX_CLONE_CHARS)
+
+ try:
+ import librosa
+
+ samples, _ = librosa.load(source, sr=24_000, mono=True)
+ limit = int(ENROLL_SECONDS * 24_000)
+ if samples.size > limit:
+ samples = samples[:limit]
+
+ try:
+ profile = enroller.enroll(samples, 24_000, name="your voice")
+ except ValueError as exc:
+ # The library's own messages name the bound and describe a good
+ # input, which is more useful than anything restated here.
+ raise gr.Error(str(exc)) from exc
+
+ # `enroll` writes no language, so every cloned voice would claim English
+ # and read its text through the English funnel.
+ profile = dataclasses.replace(profile, language=language or "en")
+
+ result = engine.synthesize_long(
+ text, profile, seed=int(seed), language=language or None, speed=float(speed)
+ )
+ return _write(result, voice="cloned", language=profile.language), _stats(result)
+ finally:
+ # Nothing the visitor recorded outlives the request that carried it.
+ with contextlib.suppress(OSError):
+ os.unlink(source)
+
+
+# --------------------------------------------------------------------------
+# Determinism probe. GPU. Renders the same text twice at the same seed.
+# --------------------------------------------------------------------------
+
+
+def _probe_duration(text, name, seed):
+ return _estimate(text, passes=2, overhead=20.0)
+
+
+@spaces.GPU(duration=_probe_duration)
+def probe(text: str, name: str, seed: float):
+ text = _check(text, MAX_PROBE_CHARS)
+ profile = PROFILES[name]
+ first = engine.synthesize_long(text, profile, seed=int(seed))
+ second = engine.synthesize_long(text, profile, seed=int(seed))
+
+ left, right = _sha256_audio(first.audio), _sha256_audio(second.audio)
+ verdict = "Identical." if left == right else "Different. Please report this."
+
+ return "\n".join(
+ [
+ f"**{verdict}**",
+ "",
+ "```",
+ f"render 1 sha256 {left}",
+ f"render 2 sha256 {right}",
+ f" algo[{first.algorithm_fingerprint}] seed {int(seed)}",
+ "```",
+ "",
+ "Identical within this build and this device. loudkit promises a "
+ "bit-identical waveform for the same seed, build, backend and input. "
+ "It does not promise that your laptop matches this GPU. "
+ f"[Read the identity contract]({IDENTITY_CONTRACT}).",
+ ]
+ )
+
+
+# --------------------------------------------------------------------------
+# Interface
+# --------------------------------------------------------------------------
+
+# loudreader.io: cream ground, ink text, black pill buttons at 14px.
+CSS = """
+#lk-head h1 { font-size: 2.15rem; margin-bottom: .25rem; letter-spacing: -.02em; }
+#lk-head p { margin-top: 0; }
+.lk-pill {
+ display: inline-block; padding: .2rem .75rem; margin: .15rem .35rem .15rem 0;
+ border: 1px solid #ded8ce; border-radius: 999px; font-size: .8rem;
+ color: #374151; background: #fffdfa;
+}
+.lk-card { background: #fffdfa; border: 1px solid #e7e1d7; border-radius: 14px; padding: .35rem 1rem; }
+footer { display: none !important; }
+"""
+
+# Gradio follows the visitor's system theme unless told otherwise, and this
+# palette is light-first. Without this the ink-on-cream tokens below land under
+# a dark stylesheet and the text turns near-white on a cream ground.
+FORCE_LIGHT = """
+() => {
+ const url = new URL(window.location);
+ if (url.searchParams.get('__theme') !== 'light') {
+ url.searchParams.set('__theme', 'light');
+ window.location.replace(url.href);
+ }
+}
+"""
+
+THEME = gr.themes.Soft(
+ primary_hue=gr.themes.colors.gray,
+ neutral_hue=gr.themes.colors.stone,
+ font=[gr.themes.GoogleFont("Inter"), "system-ui", "sans-serif"],
+).set(
+ body_background_fill="#f7f5f2",
+ body_text_color="#111827",
+ body_text_color_subdued="#4b5563",
+ block_background_fill="#fffdfa",
+ block_border_color="#e7e1d7",
+ border_color_primary="#e7e1d7",
+ input_background_fill="#ffffff",
+ button_primary_background_fill="#111827",
+ button_primary_background_fill_hover="#374151",
+ button_primary_text_color="#ffffff",
+ button_large_radius="14px",
+ button_small_radius="14px",
+)
+
+with gr.Blocks(title="loudkit", theme=THEME, css=CSS, js=FORCE_LIGHT, fill_width=False) as demo:
+ gr.Markdown(
+ f"""
+# Twenty voices. Ten languages. One engine.
+
+On-device text to speech, running here on ZeroGPU.
+[Model](https://huggingface.co/{REPO}) · [Code]({DOCS}) · [Responsible use]({RESPONSIBLE_USE})
+
+Listening costs no GPU
+Speaking and cloning spend your daily quota
+algo[{FINGERPRINT}]
+""",
+ elem_id="lk-head",
+ )
+
+ with gr.Tabs():
+ # ---------------- Listen ----------------
+ with gr.Tab("Listen"):
+ gr.Markdown(
+ "Twenty voices, rendered ahead of time and served as files. "
+ "This tab uses no GPU and spends none of your quota. "
+ "Each voice is paired with the reference recording it was enrolled from."
+ )
+ with gr.Row():
+ with gr.Column(scale=1):
+ pick = gr.Dropdown(
+ VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True
+ )
+ made = gr.Audio(label="loudkit", type="filepath", interactive=False)
+ ref = gr.Audio(label="Reference recording", type="filepath", interactive=False)
+ with gr.Column(scale=1):
+ card = gr.Markdown(elem_classes="lk-card")
+
+ with gr.Accordion("The whole roster", open=False):
+ gr.Dataframe(
+ value=ROSTER_TABLE,
+ headers=["Voice", "Language", "Gender", "Licence", "Similarity"],
+ interactive=False,
+ wrap=True,
+ )
+
+ pick.change(listen, pick, [made, ref, card])
+ demo.load(listen, pick, [made, ref, card])
+
+ # ---------------- Speak ----------------
+ with gr.Tab("Speak"):
+ gr.Markdown(
+ f"Your text, in one of the twenty voices. "
+ f"Up to {MAX_CHARS:,} characters here. The library itself takes 10,000. "
+ "This tab spends your ZeroGPU quota."
+ )
+ with gr.Row():
+ with gr.Column(scale=3):
+ say = gr.Textbox(
+ label="Text",
+ placeholder="Hello from loudkit.",
+ lines=4,
+ max_length=MAX_CHARS,
+ )
+ with gr.Column(scale=2):
+ say_voice = gr.Dropdown(
+ VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True
+ )
+ say_lang = gr.Dropdown(
+ LANGUAGE_CHOICES, value="", label="Read the text as"
+ )
+ with gr.Row():
+ say_seed = gr.Number(value=7, precision=0, label="Seed")
+ say_speed = gr.Slider(
+ lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed"
+ )
+ say_go = gr.Button("Speak", variant="primary")
+ say_out = gr.Audio(label="Speech", type="filepath")
+ say_stats = gr.Markdown()
+
+ say_go.click(
+ speak,
+ [say, say_voice, say_lang, say_seed, say_speed],
+ [say_out, say_stats],
+ )
+
+ with gr.Accordion("Determinism check", open=False):
+ gr.Markdown(
+ "This renders the same text twice at the same seed and hashes "
+ "both waveforms. The digests must match."
+ )
+ with gr.Row():
+ probe_text = gr.Textbox(
+ value="The same seed gives the same audio.",
+ label="Text",
+ max_length=MAX_PROBE_CHARS,
+ scale=3,
+ )
+ probe_voice = gr.Dropdown(
+ VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", scale=2
+ )
+ probe_seed = gr.Number(value=7, precision=0, label="Seed", scale=1)
+ probe_go = gr.Button("Render twice")
+ probe_out = gr.Markdown()
+ probe_go.click(probe, [probe_text, probe_voice, probe_seed], probe_out)
+
+ # ---------------- Clone ----------------
+ with gr.Tab("Clone"):
+ gr.Markdown(
+ f"""
+Clone a voice from a short recording, then speak with it.
+
+- Record 5 to 10 seconds. Read anything. Speak normally.
+- Clone only your own voice, or a voice you have permission to use.
+- Nothing you record is kept. The recording is deleted when the request ends.
+- See [Responsible use]({RESPONSIBLE_USE}).
+"""
+ )
+ with gr.Row():
+ with gr.Column(scale=1):
+ mic = gr.Audio(
+ sources=["microphone"],
+ type="filepath",
+ label="Record yourself",
+ )
+ with gr.Accordion("Upload a file instead", open=False):
+ upload = gr.Audio(
+ sources=["upload"], type="filepath", label="Audio file"
+ )
+ consent = gr.Checkbox(
+ value=False,
+ label=(
+ "This is my own voice, or I have permission from the "
+ "person who owns it."
+ ),
+ )
+ with gr.Column(scale=1):
+ clone_text = gr.Textbox(
+ label="Text to speak",
+ placeholder="Now in my own voice.",
+ lines=3,
+ max_length=MAX_CLONE_CHARS,
+ )
+ clone_lang = gr.Dropdown(
+ LANGUAGE_CHOICES[1:], value="en", label="Language of the text"
+ )
+ with gr.Row():
+ clone_seed = gr.Number(value=7, precision=0, label="Seed")
+ clone_speed = gr.Slider(
+ lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed"
+ )
+ clone_go = gr.Button("Clone and speak", variant="primary")
+ clone_out = gr.Audio(label="Speech", type="filepath")
+ clone_stats = gr.Markdown()
+
+ clone_go.click(
+ clone,
+ [mic, upload, consent, clone_text, clone_lang, clone_seed, clone_speed],
+ [clone_out, clone_stats],
+ )
+
+ gr.Markdown(
+ f"""
+---
+Run the same engine locally, where nothing is queued and nothing is metered.
+
+```bash
+pip install "loudkit[torch,audio,enroll,hub]"
+```
+
+```python
+import loudkit as lk
+
+engine = lk.load("{REPO}")
+voice = lk.voice("kathleen", repo="{REPO}")
+engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav")
+```
+
+Output files carry C2PA provenance: the fingerprint, the recipe and the seed.
+"""
+ )
+
+# The engine holds one set of weights and renders with an internal producer
+# thread. One render at a time keeps two requests off the same buffers.
+demo.queue(default_concurrency_limit=1, max_size=24)
+
+if __name__ == "__main__":
+ demo.launch()
diff --git a/hf-loudkit/audio/carmen.opus b/hf-loudkit/audio/carmen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..e1e18609163ff3cd67315d85497818f104d33a58
Binary files /dev/null and b/hf-loudkit/audio/carmen.opus differ
diff --git a/hf-loudkit/audio/colette.opus b/hf-loudkit/audio/colette.opus
new file mode 100644
index 0000000000000000000000000000000000000000..682389d876f822ca755d906f05db376a18ff68d9
Binary files /dev/null and b/hf-loudkit/audio/colette.opus differ
diff --git a/hf-loudkit/audio/dante.opus b/hf-loudkit/audio/dante.opus
new file mode 100644
index 0000000000000000000000000000000000000000..6628f6858d4f71d379c47411baee9507f8192928
Binary files /dev/null and b/hf-loudkit/audio/dante.opus differ
diff --git a/hf-loudkit/audio/darkman.opus b/hf-loudkit/audio/darkman.opus
new file mode 100644
index 0000000000000000000000000000000000000000..7a6ce2e7c75e4206475b532df2fe8a9850a29929
Binary files /dev/null and b/hf-loudkit/audio/darkman.opus differ
diff --git a/hf-loudkit/audio/dave.opus b/hf-loudkit/audio/dave.opus
new file mode 100644
index 0000000000000000000000000000000000000000..502d2da2e6fdf140c6310a3336457eb210a7bc31
Binary files /dev/null and b/hf-loudkit/audio/dave.opus differ
diff --git a/hf-loudkit/audio/freja.opus b/hf-loudkit/audio/freja.opus
new file mode 100644
index 0000000000000000000000000000000000000000..2160a36de61672adf91284b9c74c692b59f07206
Binary files /dev/null and b/hf-loudkit/audio/freja.opus differ
diff --git a/hf-loudkit/audio/gosia.opus b/hf-loudkit/audio/gosia.opus
new file mode 100644
index 0000000000000000000000000000000000000000..853a342df5107a4894770834ef16a8539ecf1f6f
Binary files /dev/null and b/hf-loudkit/audio/gosia.opus differ
diff --git a/hf-loudkit/audio/henri.opus b/hf-loudkit/audio/henri.opus
new file mode 100644
index 0000000000000000000000000000000000000000..97c0f1a5a9bd18e1516d97a70f26fe3a5b3ae0bb
Binary files /dev/null and b/hf-loudkit/audio/henri.opus differ
diff --git a/hf-loudkit/audio/ines.opus b/hf-loudkit/audio/ines.opus
new file mode 100644
index 0000000000000000000000000000000000000000..2f988d7adb9dc0bd7041089ac470d0cbee02c832
Binary files /dev/null and b/hf-loudkit/audio/ines.opus differ
diff --git a/hf-loudkit/audio/joe.opus b/hf-loudkit/audio/joe.opus
new file mode 100644
index 0000000000000000000000000000000000000000..d35624d4b0cd92eb44d278737e92a65c95db9892
Binary files /dev/null and b/hf-loudkit/audio/joe.opus differ
diff --git a/hf-loudkit/audio/kathleen.opus b/hf-loudkit/audio/kathleen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..afab998420c69de122c666484f135471301633ef
Binary files /dev/null and b/hf-loudkit/audio/kathleen.opus differ
diff --git a/hf-loudkit/audio/kerstin.opus b/hf-loudkit/audio/kerstin.opus
new file mode 100644
index 0000000000000000000000000000000000000000..bc9b99ead23656b1e5f2ca7af21146bc1923b6ef
Binary files /dev/null and b/hf-loudkit/audio/kerstin.opus differ
diff --git a/hf-loudkit/audio/nathalie.opus b/hf-loudkit/audio/nathalie.opus
new file mode 100644
index 0000000000000000000000000000000000000000..78d74b015d59796e31d798558ffe6c846d4e5619
Binary files /dev/null and b/hf-loudkit/audio/nathalie.opus differ
diff --git a/hf-loudkit/audio/nils.opus b/hf-loudkit/audio/nils.opus
new file mode 100644
index 0000000000000000000000000000000000000000..9f317ca8b1aba3b9aa3dd0e86719cda67125b6d8
Binary files /dev/null and b/hf-loudkit/audio/nils.opus differ
diff --git a/hf-loudkit/audio/paola.opus b/hf-loudkit/audio/paola.opus
new file mode 100644
index 0000000000000000000000000000000000000000..6f936ee395156750e75bbb580eb239a62b32eb3a
Binary files /dev/null and b/hf-loudkit/audio/paola.opus differ
diff --git a/hf-loudkit/audio/pim.opus b/hf-loudkit/audio/pim.opus
new file mode 100644
index 0000000000000000000000000000000000000000..5d31d74acd05d2cc5a8b191d3c78f06387e1eb05
Binary files /dev/null and b/hf-loudkit/audio/pim.opus differ
diff --git a/hf-loudkit/audio/refs/carmen.opus b/hf-loudkit/audio/refs/carmen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..220d1a403a3b535d284ce5c4008e3a5fcb736945
Binary files /dev/null and b/hf-loudkit/audio/refs/carmen.opus differ
diff --git a/hf-loudkit/audio/refs/colette.opus b/hf-loudkit/audio/refs/colette.opus
new file mode 100644
index 0000000000000000000000000000000000000000..dce18d4f09e5b07e10f9c46441fca118b4d17b67
Binary files /dev/null and b/hf-loudkit/audio/refs/colette.opus differ
diff --git a/hf-loudkit/audio/refs/dante.opus b/hf-loudkit/audio/refs/dante.opus
new file mode 100644
index 0000000000000000000000000000000000000000..b58368afe0db03b2d2f960c75d91280146d5f901
Binary files /dev/null and b/hf-loudkit/audio/refs/dante.opus differ
diff --git a/hf-loudkit/audio/refs/darkman.opus b/hf-loudkit/audio/refs/darkman.opus
new file mode 100644
index 0000000000000000000000000000000000000000..cb98a474936dbbdb79b3020552e955242f61ef0d
Binary files /dev/null and b/hf-loudkit/audio/refs/darkman.opus differ
diff --git a/hf-loudkit/audio/refs/dave.opus b/hf-loudkit/audio/refs/dave.opus
new file mode 100644
index 0000000000000000000000000000000000000000..ca3c66e199e7bd900d1d4180aa91b65aaba5d9e8
Binary files /dev/null and b/hf-loudkit/audio/refs/dave.opus differ
diff --git a/hf-loudkit/audio/refs/freja.opus b/hf-loudkit/audio/refs/freja.opus
new file mode 100644
index 0000000000000000000000000000000000000000..adbb18927dc20821a83eb11190e9faa2aa50efb3
Binary files /dev/null and b/hf-loudkit/audio/refs/freja.opus differ
diff --git a/hf-loudkit/audio/refs/gosia.opus b/hf-loudkit/audio/refs/gosia.opus
new file mode 100644
index 0000000000000000000000000000000000000000..ac8e283877021b29edb5bd8e27c296973cae3c07
Binary files /dev/null and b/hf-loudkit/audio/refs/gosia.opus differ
diff --git a/hf-loudkit/audio/refs/henri.opus b/hf-loudkit/audio/refs/henri.opus
new file mode 100644
index 0000000000000000000000000000000000000000..59d6873d132f1a4b61081a27cb1817fa1d463bbe
Binary files /dev/null and b/hf-loudkit/audio/refs/henri.opus differ
diff --git a/hf-loudkit/audio/refs/ines.opus b/hf-loudkit/audio/refs/ines.opus
new file mode 100644
index 0000000000000000000000000000000000000000..b31012129731dabfcdd0829dabb54a8e09d5914f
Binary files /dev/null and b/hf-loudkit/audio/refs/ines.opus differ
diff --git a/hf-loudkit/audio/refs/joe.opus b/hf-loudkit/audio/refs/joe.opus
new file mode 100644
index 0000000000000000000000000000000000000000..732465cbc480985e84fe73887ca79a520a33e374
Binary files /dev/null and b/hf-loudkit/audio/refs/joe.opus differ
diff --git a/hf-loudkit/audio/refs/kathleen.opus b/hf-loudkit/audio/refs/kathleen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..03fa0d9136a7614ef66b4d9627042992e6904be2
Binary files /dev/null and b/hf-loudkit/audio/refs/kathleen.opus differ
diff --git a/hf-loudkit/audio/refs/kerstin.opus b/hf-loudkit/audio/refs/kerstin.opus
new file mode 100644
index 0000000000000000000000000000000000000000..d6ca625df4b5c103120b206c35af9cc2c7b47936
Binary files /dev/null and b/hf-loudkit/audio/refs/kerstin.opus differ
diff --git a/hf-loudkit/audio/refs/nathalie.opus b/hf-loudkit/audio/refs/nathalie.opus
new file mode 100644
index 0000000000000000000000000000000000000000..5bc6b148d91e82fa1caa46ce33993fd078319262
Binary files /dev/null and b/hf-loudkit/audio/refs/nathalie.opus differ
diff --git a/hf-loudkit/audio/refs/nils.opus b/hf-loudkit/audio/refs/nils.opus
new file mode 100644
index 0000000000000000000000000000000000000000..bdfc6564eeeb9a75cf8f03f202369a2ae0604380
Binary files /dev/null and b/hf-loudkit/audio/refs/nils.opus differ
diff --git a/hf-loudkit/audio/refs/paola.opus b/hf-loudkit/audio/refs/paola.opus
new file mode 100644
index 0000000000000000000000000000000000000000..cef273a388c73c7e2e3f370ce4af34b56c954344
Binary files /dev/null and b/hf-loudkit/audio/refs/paola.opus differ
diff --git a/hf-loudkit/audio/refs/pim.opus b/hf-loudkit/audio/refs/pim.opus
new file mode 100644
index 0000000000000000000000000000000000000000..df2d2719e8e384f380a958142f3d700406ce6e61
Binary files /dev/null and b/hf-loudkit/audio/refs/pim.opus differ
diff --git a/hf-loudkit/audio/refs/selma.opus b/hf-loudkit/audio/refs/selma.opus
new file mode 100644
index 0000000000000000000000000000000000000000..d6bfaab9d0c0da5e7fe91eb4d2e799e388c21c89
Binary files /dev/null and b/hf-loudkit/audio/refs/selma.opus differ
diff --git a/hf-loudkit/audio/refs/soren.opus b/hf-loudkit/audio/refs/soren.opus
new file mode 100644
index 0000000000000000000000000000000000000000..dda8de236b964dbe5fd8b28e1311cb7fe65540d8
Binary files /dev/null and b/hf-loudkit/audio/refs/soren.opus differ
diff --git a/hf-loudkit/audio/refs/thorsten.opus b/hf-loudkit/audio/refs/thorsten.opus
new file mode 100644
index 0000000000000000000000000000000000000000..5e89b94fbd091c56363285eeebc1fe81deedccac
Binary files /dev/null and b/hf-loudkit/audio/refs/thorsten.opus differ
diff --git a/hf-loudkit/audio/refs/tugao.opus b/hf-loudkit/audio/refs/tugao.opus
new file mode 100644
index 0000000000000000000000000000000000000000..3df5f5af70b92a74f4a006e1a75581f82d027328
Binary files /dev/null and b/hf-loudkit/audio/refs/tugao.opus differ
diff --git a/hf-loudkit/audio/selma.opus b/hf-loudkit/audio/selma.opus
new file mode 100644
index 0000000000000000000000000000000000000000..635b80f10f7832625903ac39d8bf0e831df43eb8
Binary files /dev/null and b/hf-loudkit/audio/selma.opus differ
diff --git a/hf-loudkit/audio/soren.opus b/hf-loudkit/audio/soren.opus
new file mode 100644
index 0000000000000000000000000000000000000000..2ed33a17736947e4ed72c1b4c5fc05b93cc548a9
Binary files /dev/null and b/hf-loudkit/audio/soren.opus differ
diff --git a/hf-loudkit/audio/thorsten.opus b/hf-loudkit/audio/thorsten.opus
new file mode 100644
index 0000000000000000000000000000000000000000..278cbffa5785f98db6cd5fa1394f1eff47e6f5ef
Binary files /dev/null and b/hf-loudkit/audio/thorsten.opus differ
diff --git a/hf-loudkit/audio/tugao.opus b/hf-loudkit/audio/tugao.opus
new file mode 100644
index 0000000000000000000000000000000000000000..71f242f67742e53eecfaeb6363a4779696d3d48c
Binary files /dev/null and b/hf-loudkit/audio/tugao.opus differ
diff --git a/hf-loudkit/hf-loudkit/.gitattributes b/hf-loudkit/hf-loudkit/.gitattributes
new file mode 100644
index 0000000000000000000000000000000000000000..a6344aac8c09253b3b630fb776ae94478aa0275b
--- /dev/null
+++ b/hf-loudkit/hf-loudkit/.gitattributes
@@ -0,0 +1,35 @@
+*.7z filter=lfs diff=lfs merge=lfs -text
+*.arrow filter=lfs diff=lfs merge=lfs -text
+*.bin filter=lfs diff=lfs merge=lfs -text
+*.bz2 filter=lfs diff=lfs merge=lfs -text
+*.ckpt filter=lfs diff=lfs merge=lfs -text
+*.ftz filter=lfs diff=lfs merge=lfs -text
+*.gz filter=lfs diff=lfs merge=lfs -text
+*.h5 filter=lfs diff=lfs merge=lfs -text
+*.joblib filter=lfs diff=lfs merge=lfs -text
+*.lfs.* filter=lfs diff=lfs merge=lfs -text
+*.mlmodel filter=lfs diff=lfs merge=lfs -text
+*.model filter=lfs diff=lfs merge=lfs -text
+*.msgpack filter=lfs diff=lfs merge=lfs -text
+*.npy filter=lfs diff=lfs merge=lfs -text
+*.npz filter=lfs diff=lfs merge=lfs -text
+*.onnx filter=lfs diff=lfs merge=lfs -text
+*.ot filter=lfs diff=lfs merge=lfs -text
+*.parquet filter=lfs diff=lfs merge=lfs -text
+*.pb filter=lfs diff=lfs merge=lfs -text
+*.pickle filter=lfs diff=lfs merge=lfs -text
+*.pkl filter=lfs diff=lfs merge=lfs -text
+*.pt filter=lfs diff=lfs merge=lfs -text
+*.pth filter=lfs diff=lfs merge=lfs -text
+*.rar filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+saved_model/**/* filter=lfs diff=lfs merge=lfs -text
+*.tar.* filter=lfs diff=lfs merge=lfs -text
+*.tar filter=lfs diff=lfs merge=lfs -text
+*.tflite filter=lfs diff=lfs merge=lfs -text
+*.tgz filter=lfs diff=lfs merge=lfs -text
+*.wasm filter=lfs diff=lfs merge=lfs -text
+*.xz filter=lfs diff=lfs merge=lfs -text
+*.zip filter=lfs diff=lfs merge=lfs -text
+*.zst filter=lfs diff=lfs merge=lfs -text
+*tfevents* filter=lfs diff=lfs merge=lfs -text
diff --git a/hf-loudkit/hf-loudkit/README.md b/hf-loudkit/hf-loudkit/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..4db28afa9528f68415538245009b70427e84e4f0
--- /dev/null
+++ b/hf-loudkit/hf-loudkit/README.md
@@ -0,0 +1,77 @@
+---
+title: loudkit
+emoji: 🔊
+colorFrom: gray
+colorTo: red
+sdk: gradio
+sdk_version: 5.50.0
+python_version: "3.12.12"
+app_file: app.py
+pinned: false
+license: apache-2.0
+short_description: On-device TTS. Twenty voices, ten languages, one engine.
+models:
+ - loudreader/loudr-1
+preload_from_hub:
+ - loudreader/loudr-1 loudr-1.safetensors,loudr-1-enrollment.safetensors,ve.safetensors,manifest.json,release.json,SHA256SUMS,voices/carmen.safetensors,voices/colette.safetensors,voices/dante.safetensors,voices/darkman.safetensors,voices/dave.safetensors,voices/freja.safetensors,voices/gosia.safetensors,voices/henri.safetensors,voices/ines.safetensors,voices/joe.safetensors,voices/kathleen.safetensors,voices/kerstin.safetensors,voices/nathalie.safetensors,voices/nils.safetensors,voices/paola.safetensors,voices/pim.safetensors,voices/selma.safetensors,voices/soren.safetensors,voices/thorsten.safetensors,voices/tugao.safetensors
+---
+
+# loudkit
+
+Twenty voices across ten languages, from [loudreader/loudr-1](https://huggingface.co/loudreader/loudr-1),
+running the [loudkit](https://github.com/loudreader/loudkit) engine on ZeroGPU.
+
+## Three tabs
+
+- **Listen.** Twenty voices, each beside the reference recording it was enrolled
+ from. These files were rendered ahead of time and ship in this repo. This tab
+ uses no GPU.
+- **Speak.** Your text, in any of the twenty voices. Up to 1,000 characters.
+- **Clone.** Your own voice, from about ten seconds of audio.
+
+ZeroGPU bills GPU time to the visitor, not to the owner. An anonymous visitor
+gets about two minutes a day. A signed-in free account gets about five. Listening
+costs none of it.
+
+## Cloning and consent
+
+Clone your own voice, or a voice you have permission to use.
+
+- The microphone is the default path.
+- An upload is secondary, and needs an explicit confirmation.
+- Recordings are deleted when the request ends. Nothing is kept.
+
+See [RESPONSIBLE_USE](https://github.com/loudreader/loudkit/blob/main/RESPONSIBLE_USE.md).
+
+## Determinism
+
+The Speak tab has a determinism check. It renders the same text twice at the same
+seed and prints the SHA-256 of both waveforms. They match.
+
+That holds within this build and this device. loudkit promises a bit-identical
+waveform for the same seed, build, backend and input. It does not promise that
+your machine matches this GPU. See the
+[identity contract](https://github.com/loudreader/loudkit/blob/main/docs/reference/IDENTITY-CONTRACT.md).
+
+## Run it locally
+
+```bash
+pip install "loudkit[torch,audio,enroll,hub]"
+```
+
+```python
+import loudkit as lk
+
+engine = lk.load("loudreader/loudr-1")
+voice = lk.voice("kathleen", repo="loudreader/loudr-1")
+engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav")
+```
+
+Audio in this Space, and from `Result.save`, carries C2PA provenance: the
+algorithm fingerprint, the recipe and the seed.
+
+## Voice sources
+
+Every voice is enrolled from a public-domain or openly licensed recording.
+`voices.json` in this repo carries the full record for each one: donor, source,
+licence, consent, and the SHA-256 of both the reference and the sample.
diff --git a/hf-loudkit/hf-loudkit/app.py b/hf-loudkit/hf-loudkit/app.py
new file mode 100644
index 0000000000000000000000000000000000000000..3e2e44e07163d6b6d66f366ce80678012a99f0f9
--- /dev/null
+++ b/hf-loudkit/hf-loudkit/app.py
@@ -0,0 +1,541 @@
+"""The loudkit demo Space: hear twenty voices, speak your own text, clone your own.
+
+ZeroGPU bills GPU time to the *visitor*, not to the owner: an anonymous visitor
+gets about two minutes a day, a signed-in free account about five. A demo whose
+first click spends that budget is one most people bounce off before they have
+heard anything at all. So the Listen tab is twenty pre-rendered files served
+straight out of this repo — no GPU, no queue, no quota — and the GPU is spent
+only on what a visitor types or records.
+
+The engine and the enroller are both built on `cuda` at module level, which is
+what ZeroGPU asks for: CUDA transfers are optimised for start-up placement, and
+lazy-loading inside a `@spaces.GPU` function is explicitly discouraged. Each
+decorated call then runs in a freshly forked, short-lived process, which is also
+why there is no `torch.compile` and no CUDA graph capture here: both pay their
+cost once per process and would never amortise.
+
+Cloning is exposed, which the CPU scaffold this replaces deliberately did not do.
+The reasoning that kept it out was about consent, not about capability, so the
+consent is built into the shape of the tab rather than written beside it: the
+microphone is the default path, an upload is secondary and gated on an explicit
+confirmation, and neither recording outlives the request that carried it.
+"""
+
+from __future__ import annotations
+
+import contextlib
+import dataclasses
+import hashlib
+import json
+import os
+import tempfile
+from pathlib import Path
+
+# Before torch, and before anything that imports torch. The module installs the
+# CUDA emulation that lets a module-level `.to("cuda")` succeed on a machine
+# that has no GPU attached yet.
+import spaces
+
+import gradio as gr
+import numpy as np
+
+import loudkit as lk
+from loudkit.backends.torch_backend import build_torch_enroller
+from loudkit.hub import resolve_enrollment_checkpoint, resolve_voice_encoder
+
+REPO = "loudreader/loudr-1"
+DEVICE = "cuda"
+HERE = Path(__file__).parent
+
+# The CPU scaffold capped text at 300 characters because CPU synthesis ran at
+# roughly a tenth of real time. On a GPU the cap is about the visitor's daily
+# quota instead, which is a far looser bound: 1000 characters is ~70 s of speech.
+MAX_CHARS = 1_000
+MAX_CLONE_CHARS = 400
+MAX_PROBE_CHARS = 200
+
+# The enroller refuses anything over 30 s and wants 5 to 10. The prompt is built
+# from the first 10 s; the speaker embedding reads whatever else is there, so a
+# little past the prompt window is useful and 20 s stays clear of the refusal.
+ENROLL_SECONDS = 20.0
+
+DOCS = "https://github.com/loudreader/loudkit"
+IDENTITY_CONTRACT = f"{DOCS}/blob/main/docs/reference/IDENTITY-CONTRACT.md"
+RESPONSIBLE_USE = f"{DOCS}/blob/main/RESPONSIBLE_USE.md"
+
+ROSTER = json.loads((HERE / "voices.json").read_text(encoding="utf-8"))
+BY_NAME = {entry["name"]: entry for entry in ROSTER}
+ORDERED = sorted(ROSTER, key=lambda e: (e["language"], e["name"]))
+VOICE_CHOICES = [(f"{e['name']} · {e['language']} ({e['gender']})", e["name"]) for e in ORDERED]
+
+# --------------------------------------------------------------------------
+# Module-level model placement, per the ZeroGPU contract.
+# --------------------------------------------------------------------------
+
+engine = lk.load(REPO, device=DEVICE)
+
+# Voice profiles are numpy, not torch, so they are device-agnostic and cost a
+# few hundred kilobytes each. Loading all twenty up front means switching voice
+# in the Speak tab never blocks on a download.
+PROFILES = {entry["name"]: lk.voice(entry["name"], repo=REPO) for entry in ROSTER}
+
+# Enrollment reads the other half of the release: the speech tokenizer and the
+# speaker encoder, which synthesis never touches, plus the utterance voice
+# encoder that sits beside both. `lk.enroll()` builds this per call by design;
+# a Space would pay the load on every clone, so it is built once here instead.
+enroller = build_torch_enroller(
+ str(resolve_enrollment_checkpoint(REPO)),
+ device=DEVICE,
+ voice_encoder_weights=str(resolve_voice_encoder(REPO)),
+)
+
+FINGERPRINT = engine.algorithm.fingerprint()
+
+_LANGUAGE_NAMES = {e["language_id"]: e["language"] for e in ROSTER}
+LANGUAGE_CHOICES = [("Follow the voice", "")] + [
+ (f"{_LANGUAGE_NAMES.get(code, code)} ({code})", code) for code in lk.languages()
+]
+
+# --------------------------------------------------------------------------
+# Helpers
+# --------------------------------------------------------------------------
+
+
+def _sha256_audio(audio: np.ndarray) -> str:
+ """Hash the waveform, not the file.
+
+ `Result.save` appends a C2PA manifest carrying a wall-clock creation time,
+ which the library itself calls the one byte range in which two identical
+ renders may legitimately differ. Hashing the saved WAV would therefore print
+ two different digests for two identical renders and read as a determinism
+ failure. The waveform is what the identity contract makes its promise about,
+ so the waveform is what gets hashed.
+ """
+ return hashlib.sha256(np.ascontiguousarray(audio, dtype=np.float32).tobytes()).hexdigest()
+
+
+def _write(result: lk.Result, *, voice: str, language: str) -> str:
+ out = tempfile.NamedTemporaryFile(suffix=".wav", delete=False)
+ out.close()
+ # Provenance on: the manifest carries the fingerprint, the recipe and the
+ # seed, which is the machine-readable marking a synthetic-speech demo should
+ # be handing out by default.
+ result.save(out.name, voice=voice, language=language)
+ return out.name
+
+
+def _stats(result: lk.Result) -> str:
+ seconds = len(result.audio) / result.sample_rate
+ return (
+ f"**{seconds:.1f} s of audio.** {result.timings.describe(seconds)}\n\n"
+ f"`algo[{result.algorithm_fingerprint}]` · seed `{result.seed}` · "
+ f"speed `{result.speed:g}x` · {result.sample_rate} Hz"
+ )
+
+
+def _estimate(text: str, *, passes: int = 1, overhead: float = 15.0) -> int:
+ """Seconds of GPU to ask for.
+
+ Speech runs at roughly 14 characters a second, and the render is asked to
+ keep up with better than real time; the overhead covers the process fork and
+ the first real CUDA touch. Asking for too much costs queue priority but not
+ quota, which is charged on effective duration, so this leans generous.
+ """
+ audio_seconds = len((text or "").strip()) / 14.0
+ return int(min(180.0, overhead + passes * max(4.0, audio_seconds * 0.9)))
+
+
+def _check(text: str, limit: int) -> str:
+ text = (text or "").strip()
+ if not text:
+ raise gr.Error("Type something to say.")
+ if len(text) > limit:
+ raise gr.Error(f"Keep it under {limit:,} characters here. The library itself takes 10,000.")
+ return text
+
+
+# --------------------------------------------------------------------------
+# Listen. No GPU: these files were rendered ahead of time and ship in the repo.
+# --------------------------------------------------------------------------
+
+
+def listen(name: str):
+ entry = BY_NAME[name]
+ sample, reference, source = entry["sample"], entry["reference"], entry["source"]
+
+ lines = [
+ f"### {entry['name']}. {entry['language']} ({entry['gender']}).",
+ "",
+ f"> {sample['text']}",
+ "",
+ f"From *{sample['work']}*, seed `{sample['seed']}`.",
+ "",
+ f"- Reference recording: {reference['duration_s']:.1f} s, {reference['construction']}.",
+ f"- Source: [{source['name']}]({source['url']}), {source['license']}.",
+ f"- Consent: {source['consent']}.",
+ ]
+ similarity = entry.get("speaker_similarity")
+ if similarity is not None:
+ lines.append(f"- Speaker similarity to the reference: {similarity:.3f}.")
+ lines.append(f"- Voice profile: `{entry['profile']['hf_path']}`.")
+
+ return (
+ str(HERE / sample["audio"]),
+ str(HERE / reference["public_preview"]),
+ "\n".join(lines),
+ )
+
+
+ROSTER_TABLE = [
+ [
+ entry["name"],
+ entry["language"],
+ entry["gender"],
+ entry["source"]["license"],
+ f"{entry['speaker_similarity']:.3f}" if entry.get("speaker_similarity") is not None else "",
+ ]
+ for entry in ORDERED
+]
+
+
+# --------------------------------------------------------------------------
+# Speak. GPU.
+# --------------------------------------------------------------------------
+
+
+def _speak_duration(text, name, language, seed, speed):
+ return _estimate(text, overhead=15.0)
+
+
+@spaces.GPU(duration=_speak_duration)
+def speak(text: str, name: str, language: str, seed: float, speed: float):
+ text = _check(text, MAX_CHARS)
+ result = engine.synthesize_long(
+ text,
+ PROFILES[name],
+ seed=int(seed),
+ language=language or None,
+ speed=float(speed),
+ )
+ label = language or BY_NAME[name]["language_id"]
+ return _write(result, voice=name, language=label), _stats(result)
+
+
+# --------------------------------------------------------------------------
+# Clone. GPU. The microphone is the default path; an upload is gated.
+# --------------------------------------------------------------------------
+
+
+def _clone_duration(mic, upload, consent, text, language, seed, speed):
+ # Enrollment is a fixed cost on top of the render: two encoders and a
+ # tokenizer over at most 20 s of audio.
+ return _estimate(text, overhead=30.0)
+
+
+@spaces.GPU(duration=_clone_duration)
+def clone(mic, upload, consent: bool, text: str, language: str, seed: float, speed: float):
+ source = mic or upload
+ if not source:
+ raise gr.Error("Record yourself first, or upload a clip you are allowed to use.")
+ if upload and not mic and not consent:
+ raise gr.Error("Confirm the uploaded voice is yours, or that you have permission to use it.")
+ text = _check(text, MAX_CLONE_CHARS)
+
+ try:
+ import librosa
+
+ samples, _ = librosa.load(source, sr=24_000, mono=True)
+ limit = int(ENROLL_SECONDS * 24_000)
+ if samples.size > limit:
+ samples = samples[:limit]
+
+ try:
+ profile = enroller.enroll(samples, 24_000, name="your voice")
+ except ValueError as exc:
+ # The library's own messages name the bound and describe a good
+ # input, which is more useful than anything restated here.
+ raise gr.Error(str(exc)) from exc
+
+ # `enroll` writes no language, so every cloned voice would claim English
+ # and read its text through the English funnel.
+ profile = dataclasses.replace(profile, language=language or "en")
+
+ result = engine.synthesize_long(
+ text, profile, seed=int(seed), language=language or None, speed=float(speed)
+ )
+ return _write(result, voice="cloned", language=profile.language), _stats(result)
+ finally:
+ # Nothing the visitor recorded outlives the request that carried it.
+ with contextlib.suppress(OSError):
+ os.unlink(source)
+
+
+# --------------------------------------------------------------------------
+# Determinism probe. GPU. Renders the same text twice at the same seed.
+# --------------------------------------------------------------------------
+
+
+def _probe_duration(text, name, seed):
+ return _estimate(text, passes=2, overhead=20.0)
+
+
+@spaces.GPU(duration=_probe_duration)
+def probe(text: str, name: str, seed: float):
+ text = _check(text, MAX_PROBE_CHARS)
+ profile = PROFILES[name]
+ first = engine.synthesize_long(text, profile, seed=int(seed))
+ second = engine.synthesize_long(text, profile, seed=int(seed))
+
+ left, right = _sha256_audio(first.audio), _sha256_audio(second.audio)
+ verdict = "Identical." if left == right else "Different. Please report this."
+
+ return "\n".join(
+ [
+ f"**{verdict}**",
+ "",
+ "```",
+ f"render 1 sha256 {left}",
+ f"render 2 sha256 {right}",
+ f" algo[{first.algorithm_fingerprint}] seed {int(seed)}",
+ "```",
+ "",
+ "Identical within this build and this device. loudkit promises a "
+ "bit-identical waveform for the same seed, build, backend and input. "
+ "It does not promise that your laptop matches this GPU. "
+ f"[Read the identity contract]({IDENTITY_CONTRACT}).",
+ ]
+ )
+
+
+# --------------------------------------------------------------------------
+# Interface
+# --------------------------------------------------------------------------
+
+# loudreader.io: cream ground, ink text, black pill buttons at 14px.
+CSS = """
+#lk-head h1 { font-size: 2.15rem; margin-bottom: .25rem; letter-spacing: -.02em; }
+#lk-head p { margin-top: 0; }
+.lk-pill {
+ display: inline-block; padding: .2rem .75rem; margin: .15rem .35rem .15rem 0;
+ border: 1px solid #ded8ce; border-radius: 999px; font-size: .8rem;
+ color: #374151; background: #fffdfa;
+}
+.lk-card { background: #fffdfa; border: 1px solid #e7e1d7; border-radius: 14px; padding: .35rem 1rem; }
+footer { display: none !important; }
+"""
+
+# Gradio follows the visitor's system theme unless told otherwise, and this
+# palette is light-first. Without this the ink-on-cream tokens below land under
+# a dark stylesheet and the text turns near-white on a cream ground.
+FORCE_LIGHT = """
+() => {
+ const url = new URL(window.location);
+ if (url.searchParams.get('__theme') !== 'light') {
+ url.searchParams.set('__theme', 'light');
+ window.location.replace(url.href);
+ }
+}
+"""
+
+THEME = gr.themes.Soft(
+ primary_hue=gr.themes.colors.gray,
+ neutral_hue=gr.themes.colors.stone,
+ font=[gr.themes.GoogleFont("Inter"), "system-ui", "sans-serif"],
+).set(
+ body_background_fill="#f7f5f2",
+ body_text_color="#111827",
+ body_text_color_subdued="#4b5563",
+ block_background_fill="#fffdfa",
+ block_border_color="#e7e1d7",
+ border_color_primary="#e7e1d7",
+ input_background_fill="#ffffff",
+ button_primary_background_fill="#111827",
+ button_primary_background_fill_hover="#374151",
+ button_primary_text_color="#ffffff",
+ button_large_radius="14px",
+ button_small_radius="14px",
+)
+
+with gr.Blocks(title="loudkit", theme=THEME, css=CSS, js=FORCE_LIGHT, fill_width=False) as demo:
+ gr.Markdown(
+ f"""
+# Twenty voices. Ten languages. One engine.
+
+On-device text to speech, running here on ZeroGPU.
+[Model](https://huggingface.co/{REPO}) · [Code]({DOCS}) · [Responsible use]({RESPONSIBLE_USE})
+
+Listening costs no GPU
+Speaking and cloning spend your daily quota
+algo[{FINGERPRINT}]
+""",
+ elem_id="lk-head",
+ )
+
+ with gr.Tabs():
+ # ---------------- Listen ----------------
+ with gr.Tab("Listen"):
+ gr.Markdown(
+ "Twenty voices, rendered ahead of time and served as files. "
+ "This tab uses no GPU and spends none of your quota. "
+ "Each voice is paired with the reference recording it was enrolled from."
+ )
+ with gr.Row():
+ with gr.Column(scale=1):
+ pick = gr.Dropdown(
+ VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True
+ )
+ made = gr.Audio(label="loudkit", type="filepath", interactive=False)
+ ref = gr.Audio(label="Reference recording", type="filepath", interactive=False)
+ with gr.Column(scale=1):
+ card = gr.Markdown(elem_classes="lk-card")
+
+ with gr.Accordion("The whole roster", open=False):
+ gr.Dataframe(
+ value=ROSTER_TABLE,
+ headers=["Voice", "Language", "Gender", "Licence", "Similarity"],
+ interactive=False,
+ wrap=True,
+ )
+
+ pick.change(listen, pick, [made, ref, card])
+ demo.load(listen, pick, [made, ref, card])
+
+ # ---------------- Speak ----------------
+ with gr.Tab("Speak"):
+ gr.Markdown(
+ f"Your text, in one of the twenty voices. "
+ f"Up to {MAX_CHARS:,} characters here. The library itself takes 10,000. "
+ "This tab spends your ZeroGPU quota."
+ )
+ with gr.Row():
+ with gr.Column(scale=3):
+ say = gr.Textbox(
+ label="Text",
+ placeholder="Hello from loudkit.",
+ lines=4,
+ max_length=MAX_CHARS,
+ )
+ with gr.Column(scale=2):
+ say_voice = gr.Dropdown(
+ VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", filterable=True
+ )
+ say_lang = gr.Dropdown(
+ LANGUAGE_CHOICES, value="", label="Read the text as"
+ )
+ with gr.Row():
+ say_seed = gr.Number(value=7, precision=0, label="Seed")
+ say_speed = gr.Slider(
+ lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed"
+ )
+ say_go = gr.Button("Speak", variant="primary")
+ say_out = gr.Audio(label="Speech", type="filepath")
+ say_stats = gr.Markdown()
+
+ say_go.click(
+ speak,
+ [say, say_voice, say_lang, say_seed, say_speed],
+ [say_out, say_stats],
+ )
+
+ with gr.Accordion("Determinism check", open=False):
+ gr.Markdown(
+ "This renders the same text twice at the same seed and hashes "
+ "both waveforms. The digests must match."
+ )
+ with gr.Row():
+ probe_text = gr.Textbox(
+ value="The same seed gives the same audio.",
+ label="Text",
+ max_length=MAX_PROBE_CHARS,
+ scale=3,
+ )
+ probe_voice = gr.Dropdown(
+ VOICE_CHOICES, value=ORDERED[0]["name"], label="Voice", scale=2
+ )
+ probe_seed = gr.Number(value=7, precision=0, label="Seed", scale=1)
+ probe_go = gr.Button("Render twice")
+ probe_out = gr.Markdown()
+ probe_go.click(probe, [probe_text, probe_voice, probe_seed], probe_out)
+
+ # ---------------- Clone ----------------
+ with gr.Tab("Clone"):
+ gr.Markdown(
+ f"""
+Clone a voice from a short recording, then speak with it.
+
+- Record 5 to 10 seconds. Read anything. Speak normally.
+- Clone only your own voice, or a voice you have permission to use.
+- Nothing you record is kept. The recording is deleted when the request ends.
+- See [Responsible use]({RESPONSIBLE_USE}).
+"""
+ )
+ with gr.Row():
+ with gr.Column(scale=1):
+ mic = gr.Audio(
+ sources=["microphone"],
+ type="filepath",
+ label="Record yourself",
+ )
+ with gr.Accordion("Upload a file instead", open=False):
+ upload = gr.Audio(
+ sources=["upload"], type="filepath", label="Audio file"
+ )
+ consent = gr.Checkbox(
+ value=False,
+ label=(
+ "This is my own voice, or I have permission from the "
+ "person who owns it."
+ ),
+ )
+ with gr.Column(scale=1):
+ clone_text = gr.Textbox(
+ label="Text to speak",
+ placeholder="Now in my own voice.",
+ lines=3,
+ max_length=MAX_CLONE_CHARS,
+ )
+ clone_lang = gr.Dropdown(
+ LANGUAGE_CHOICES[1:], value="en", label="Language of the text"
+ )
+ with gr.Row():
+ clone_seed = gr.Number(value=7, precision=0, label="Seed")
+ clone_speed = gr.Slider(
+ lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed"
+ )
+ clone_go = gr.Button("Clone and speak", variant="primary")
+ clone_out = gr.Audio(label="Speech", type="filepath")
+ clone_stats = gr.Markdown()
+
+ clone_go.click(
+ clone,
+ [mic, upload, consent, clone_text, clone_lang, clone_seed, clone_speed],
+ [clone_out, clone_stats],
+ )
+
+ gr.Markdown(
+ f"""
+---
+Run the same engine locally, where nothing is queued and nothing is metered.
+
+```bash
+pip install "loudkit[torch,audio,enroll,hub]"
+```
+
+```python
+import loudkit as lk
+
+engine = lk.load("{REPO}")
+voice = lk.voice("kathleen", repo="{REPO}")
+engine.synthesize_long("Hello from loudkit.", voice, seed=7).save("hello.wav")
+```
+
+Output files carry C2PA provenance: the fingerprint, the recipe and the seed.
+"""
+ )
+
+# The engine holds one set of weights and renders with an internal producer
+# thread. One render at a time keeps two requests off the same buffers.
+demo.queue(default_concurrency_limit=1, max_size=24)
+
+if __name__ == "__main__":
+ demo.launch()
diff --git a/hf-loudkit/hf-loudkit/audio/carmen.opus b/hf-loudkit/hf-loudkit/audio/carmen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..e1e18609163ff3cd67315d85497818f104d33a58
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/carmen.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/colette.opus b/hf-loudkit/hf-loudkit/audio/colette.opus
new file mode 100644
index 0000000000000000000000000000000000000000..682389d876f822ca755d906f05db376a18ff68d9
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/colette.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/dante.opus b/hf-loudkit/hf-loudkit/audio/dante.opus
new file mode 100644
index 0000000000000000000000000000000000000000..6628f6858d4f71d379c47411baee9507f8192928
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/dante.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/darkman.opus b/hf-loudkit/hf-loudkit/audio/darkman.opus
new file mode 100644
index 0000000000000000000000000000000000000000..7a6ce2e7c75e4206475b532df2fe8a9850a29929
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/darkman.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/dave.opus b/hf-loudkit/hf-loudkit/audio/dave.opus
new file mode 100644
index 0000000000000000000000000000000000000000..502d2da2e6fdf140c6310a3336457eb210a7bc31
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/dave.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/freja.opus b/hf-loudkit/hf-loudkit/audio/freja.opus
new file mode 100644
index 0000000000000000000000000000000000000000..2160a36de61672adf91284b9c74c692b59f07206
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/freja.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/gosia.opus b/hf-loudkit/hf-loudkit/audio/gosia.opus
new file mode 100644
index 0000000000000000000000000000000000000000..853a342df5107a4894770834ef16a8539ecf1f6f
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/gosia.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/henri.opus b/hf-loudkit/hf-loudkit/audio/henri.opus
new file mode 100644
index 0000000000000000000000000000000000000000..97c0f1a5a9bd18e1516d97a70f26fe3a5b3ae0bb
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/henri.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/ines.opus b/hf-loudkit/hf-loudkit/audio/ines.opus
new file mode 100644
index 0000000000000000000000000000000000000000..2f988d7adb9dc0bd7041089ac470d0cbee02c832
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/ines.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/joe.opus b/hf-loudkit/hf-loudkit/audio/joe.opus
new file mode 100644
index 0000000000000000000000000000000000000000..d35624d4b0cd92eb44d278737e92a65c95db9892
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/joe.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/kathleen.opus b/hf-loudkit/hf-loudkit/audio/kathleen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..afab998420c69de122c666484f135471301633ef
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/kathleen.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/kerstin.opus b/hf-loudkit/hf-loudkit/audio/kerstin.opus
new file mode 100644
index 0000000000000000000000000000000000000000..bc9b99ead23656b1e5f2ca7af21146bc1923b6ef
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/kerstin.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/nathalie.opus b/hf-loudkit/hf-loudkit/audio/nathalie.opus
new file mode 100644
index 0000000000000000000000000000000000000000..78d74b015d59796e31d798558ffe6c846d4e5619
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/nathalie.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/nils.opus b/hf-loudkit/hf-loudkit/audio/nils.opus
new file mode 100644
index 0000000000000000000000000000000000000000..9f317ca8b1aba3b9aa3dd0e86719cda67125b6d8
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/nils.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/paola.opus b/hf-loudkit/hf-loudkit/audio/paola.opus
new file mode 100644
index 0000000000000000000000000000000000000000..6f936ee395156750e75bbb580eb239a62b32eb3a
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/paola.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/pim.opus b/hf-loudkit/hf-loudkit/audio/pim.opus
new file mode 100644
index 0000000000000000000000000000000000000000..5d31d74acd05d2cc5a8b191d3c78f06387e1eb05
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/pim.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/carmen.opus b/hf-loudkit/hf-loudkit/audio/refs/carmen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..220d1a403a3b535d284ce5c4008e3a5fcb736945
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/carmen.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/colette.opus b/hf-loudkit/hf-loudkit/audio/refs/colette.opus
new file mode 100644
index 0000000000000000000000000000000000000000..dce18d4f09e5b07e10f9c46441fca118b4d17b67
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/colette.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/dante.opus b/hf-loudkit/hf-loudkit/audio/refs/dante.opus
new file mode 100644
index 0000000000000000000000000000000000000000..b58368afe0db03b2d2f960c75d91280146d5f901
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/dante.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/darkman.opus b/hf-loudkit/hf-loudkit/audio/refs/darkman.opus
new file mode 100644
index 0000000000000000000000000000000000000000..cb98a474936dbbdb79b3020552e955242f61ef0d
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/darkman.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/dave.opus b/hf-loudkit/hf-loudkit/audio/refs/dave.opus
new file mode 100644
index 0000000000000000000000000000000000000000..ca3c66e199e7bd900d1d4180aa91b65aaba5d9e8
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/dave.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/freja.opus b/hf-loudkit/hf-loudkit/audio/refs/freja.opus
new file mode 100644
index 0000000000000000000000000000000000000000..adbb18927dc20821a83eb11190e9faa2aa50efb3
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/freja.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/gosia.opus b/hf-loudkit/hf-loudkit/audio/refs/gosia.opus
new file mode 100644
index 0000000000000000000000000000000000000000..ac8e283877021b29edb5bd8e27c296973cae3c07
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/gosia.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/henri.opus b/hf-loudkit/hf-loudkit/audio/refs/henri.opus
new file mode 100644
index 0000000000000000000000000000000000000000..59d6873d132f1a4b61081a27cb1817fa1d463bbe
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/henri.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/ines.opus b/hf-loudkit/hf-loudkit/audio/refs/ines.opus
new file mode 100644
index 0000000000000000000000000000000000000000..b31012129731dabfcdd0829dabb54a8e09d5914f
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/ines.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/joe.opus b/hf-loudkit/hf-loudkit/audio/refs/joe.opus
new file mode 100644
index 0000000000000000000000000000000000000000..732465cbc480985e84fe73887ca79a520a33e374
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/joe.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/kathleen.opus b/hf-loudkit/hf-loudkit/audio/refs/kathleen.opus
new file mode 100644
index 0000000000000000000000000000000000000000..03fa0d9136a7614ef66b4d9627042992e6904be2
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/kathleen.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/kerstin.opus b/hf-loudkit/hf-loudkit/audio/refs/kerstin.opus
new file mode 100644
index 0000000000000000000000000000000000000000..d6ca625df4b5c103120b206c35af9cc2c7b47936
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/kerstin.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/nathalie.opus b/hf-loudkit/hf-loudkit/audio/refs/nathalie.opus
new file mode 100644
index 0000000000000000000000000000000000000000..5bc6b148d91e82fa1caa46ce33993fd078319262
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/nathalie.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/nils.opus b/hf-loudkit/hf-loudkit/audio/refs/nils.opus
new file mode 100644
index 0000000000000000000000000000000000000000..bdfc6564eeeb9a75cf8f03f202369a2ae0604380
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/nils.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/paola.opus b/hf-loudkit/hf-loudkit/audio/refs/paola.opus
new file mode 100644
index 0000000000000000000000000000000000000000..cef273a388c73c7e2e3f370ce4af34b56c954344
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/paola.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/pim.opus b/hf-loudkit/hf-loudkit/audio/refs/pim.opus
new file mode 100644
index 0000000000000000000000000000000000000000..df2d2719e8e384f380a958142f3d700406ce6e61
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/pim.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/selma.opus b/hf-loudkit/hf-loudkit/audio/refs/selma.opus
new file mode 100644
index 0000000000000000000000000000000000000000..d6bfaab9d0c0da5e7fe91eb4d2e799e388c21c89
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/selma.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/soren.opus b/hf-loudkit/hf-loudkit/audio/refs/soren.opus
new file mode 100644
index 0000000000000000000000000000000000000000..dda8de236b964dbe5fd8b28e1311cb7fe65540d8
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/soren.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/thorsten.opus b/hf-loudkit/hf-loudkit/audio/refs/thorsten.opus
new file mode 100644
index 0000000000000000000000000000000000000000..5e89b94fbd091c56363285eeebc1fe81deedccac
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/thorsten.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/refs/tugao.opus b/hf-loudkit/hf-loudkit/audio/refs/tugao.opus
new file mode 100644
index 0000000000000000000000000000000000000000..3df5f5af70b92a74f4a006e1a75581f82d027328
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/refs/tugao.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/selma.opus b/hf-loudkit/hf-loudkit/audio/selma.opus
new file mode 100644
index 0000000000000000000000000000000000000000..635b80f10f7832625903ac39d8bf0e831df43eb8
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/selma.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/soren.opus b/hf-loudkit/hf-loudkit/audio/soren.opus
new file mode 100644
index 0000000000000000000000000000000000000000..2ed33a17736947e4ed72c1b4c5fc05b93cc548a9
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/soren.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/thorsten.opus b/hf-loudkit/hf-loudkit/audio/thorsten.opus
new file mode 100644
index 0000000000000000000000000000000000000000..278cbffa5785f98db6cd5fa1394f1eff47e6f5ef
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/thorsten.opus differ
diff --git a/hf-loudkit/hf-loudkit/audio/tugao.opus b/hf-loudkit/hf-loudkit/audio/tugao.opus
new file mode 100644
index 0000000000000000000000000000000000000000..71f242f67742e53eecfaeb6363a4779696d3d48c
Binary files /dev/null and b/hf-loudkit/hf-loudkit/audio/tugao.opus differ
diff --git a/hf-loudkit/hf-loudkit/requirements.txt b/hf-loudkit/hf-loudkit/requirements.txt
new file mode 100644
index 0000000000000000000000000000000000000000..60b4bdb741c3b04aa1bb164e82cbfb4bbb071a24
--- /dev/null
+++ b/hf-loudkit/hf-loudkit/requirements.txt
@@ -0,0 +1,18 @@
+# The library under demo. 0.1.0 is on PyPI, so this installs from the registry
+# rather than carrying a wheel beside the app.
+#
+# torch the cuda backend this Space runs on
+# audio soundfile, for Result.save
+# enroll torchaudio + librosa, for the Clone tab
+# hub huggingface_hub, to resolve loudreader/loudr-1
+loudkit[torch,audio,enroll,hub]==0.1.0
+
+# loudkit floors torch at >=2.4 and sets no ceiling, and ZeroGPU runs a fixed
+# set of builds: 2.8.0, 2.9.1, 2.10.0, 2.11.0. Pinned here so the resolver
+# cannot land outside that window. torchaudio tracks torch version for version.
+torch==2.8.0
+torchaudio==2.8.0
+
+# The ZeroGPU decorator. Present on ZeroGPU hardware already; named here so a
+# local run or a duplicated Space installs it too.
+spaces>=0.42
diff --git a/hf-loudkit/hf-loudkit/voices.json b/hf-loudkit/hf-loudkit/voices.json
new file mode 100644
index 0000000000000000000000000000000000000000..ac89f1f9159d32cd3c35686f32448d57a27284b2
--- /dev/null
+++ b/hf-loudkit/hf-loudkit/voices.json
@@ -0,0 +1,683 @@
+[
+ {
+ "name": "darkman",
+ "language": "Polish",
+ "language_id": "pl",
+ "gender": "M",
+ "donor": "darkman (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "darkman.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/darkman.opus",
+ "sha256": "d228db9d17b1ddb4ba450f16a6b8650bb0a8550521b78a914cb9e573cd24d576",
+ "duration_s": 9.78,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/darkman.safetensors",
+ "sha256": "78988a6769f90c209240f9f285eb9cdd760d1ba471cfa7d6e67db632599cfa1b"
+ },
+ "speaker_similarity": 0.935,
+ "sample": {
+ "seed": 7,
+ "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.",
+ "work": "Quo Vadis (Henryk Sienkiewicz) — public domain",
+ "audio": "audio/darkman.opus",
+ "sha256": "b30f98cf0452e9c694fcbafc27b413041c87022238f5424fc1aa452c8eb156a5"
+ }
+ },
+ {
+ "name": "gosia",
+ "language": "Polish",
+ "language_id": "pl",
+ "gender": "F",
+ "donor": "gosia (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "gosia.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/gosia.opus",
+ "sha256": "02c1d09db8b097c66dd8705747a6ef138596f5f91c3213b1cd430ae75ca2279a",
+ "duration_s": 10.86,
+ "construction": "concatenated long donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/gosia.safetensors",
+ "sha256": "e115cf64bfa0ad15770a50c0be8097312a58d41f84c7bb4f01ecbdbac5d21823"
+ },
+ "speaker_similarity": 0.917,
+ "sample": {
+ "seed": 7,
+ "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.",
+ "work": "Quo Vadis (Henryk Sienkiewicz) — public domain",
+ "audio": "audio/gosia.opus",
+ "sha256": "4ba1107d7f807ac2e0941167d1bc4a7cfb2039a4b0482a3bc798db5eeefe17e5"
+ }
+ },
+ {
+ "name": "joe",
+ "language": "English",
+ "language_id": "en",
+ "gender": "M",
+ "donor": "joe (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "joe.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/joe.opus",
+ "sha256": "591c8d6a5799a4266d029762dc0f671ac9f0455382aeef1e763ca206d6fdd164",
+ "duration_s": 10.99,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/joe.safetensors",
+ "sha256": "cfc3687fdb1f6b56fe13e252ca4a5bd681cef0702dc5cd56d4ebbc60e1aea6e8"
+ },
+ "speaker_similarity": 0.924,
+ "sample": {
+ "seed": 11,
+ "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.",
+ "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain",
+ "audio": "audio/joe.opus",
+ "sha256": "83e0bf0a1f6b03e047de66871536c3ff77291041d0a3f2907b6990e580519fea"
+ }
+ },
+ {
+ "name": "kathleen",
+ "language": "English",
+ "language_id": "en",
+ "gender": "F",
+ "donor": "kathleen (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "kathleen.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/kathleen.opus",
+ "sha256": "f43a24359a333e1f723c5f10d7dc799ec5880abf2b5b9321a7d5f8c7064d6f73",
+ "duration_s": 11.0,
+ "construction": "concatenated donation clips; prompt texts from the CMU ARCTIC prompt list (public-domain Gutenberg novels), recording is kathleen's own CC0 donation"
+ },
+ "profile": {
+ "hf_path": "voices/kathleen.safetensors",
+ "sha256": "6a845e4eb11989fc9bd8e19eed7c5ac585e6d08e1ec824211f0eb240ce523d29"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.",
+ "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain",
+ "audio": "audio/kathleen.opus",
+ "sha256": "48615517c1d1d2d55dbb7f74f37982918236e33f079947459d56c116cd70b7a3"
+ }
+ },
+ {
+ "name": "thorsten",
+ "language": "German",
+ "language_id": "de",
+ "gender": "M",
+ "donor": "Thorsten Mueller",
+ "speaker_id": null,
+ "source": {
+ "name": "Thorsten-Voice TV-44kHz-Full",
+ "url": "https://huggingface.co/datasets/Thorsten-Voice/TV-44kHz-Full",
+ "license": "CC0-1.0",
+ "consent": "voice deliberately donated by Thorsten Mueller for TTS"
+ },
+ "reference": {
+ "source_filename": "thorsten.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/thorsten.opus",
+ "sha256": "37805d1136f727f7530e714b772c23224ae1c6f6931ea5b78399cf286694de3e",
+ "duration_s": 6.37,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/thorsten.safetensors",
+ "sha256": "87873fec8d3dceaa643ab1ab07cf5143364e659dd6b3caec707f51408ac2154a"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.",
+ "work": "Max und Moritz (Wilhelm Busch) — public domain",
+ "audio": "audio/thorsten.opus",
+ "sha256": "d45938440534f7fa11afe388fa41a9a56879809a65c8477605220ed53d5a938b"
+ }
+ },
+ {
+ "name": "kerstin",
+ "language": "German",
+ "language_id": "de",
+ "gender": "F",
+ "donor": "kerstin (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "kerstin.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/kerstin.opus",
+ "sha256": "34d7c0db776bcda98e56bc130ced840debe0b12a73f4b03c556b30fc356db187",
+ "duration_s": 9.79,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/kerstin.safetensors",
+ "sha256": "d096ecde83dfc5de7a927acf67a5304c55f863ca722f826c67fafd31c48609a6"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.",
+ "work": "Max und Moritz (Wilhelm Busch) — public domain",
+ "audio": "audio/kerstin.opus",
+ "sha256": "dc1990efd85eafee75810f45fd259505d4840579f008aba9f38ca1a7c42e839b"
+ }
+ },
+ {
+ "name": "henri",
+ "language": "French",
+ "language_id": "fr",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "10087",
+ "source": {
+ "name": "Kyutai tts-voices, cml-tts french enhanced references",
+ "url": "https://huggingface.co/kyutai/tts-voices",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "henri.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/henri.opus",
+ "sha256": "9185f21daf7bcbb9a53ea06b059438ea83b7315ef4f852d69d56afc562061fc0",
+ "duration_s": 6.56,
+ "construction": "pre-cut enhanced 10 s reference"
+ },
+ "profile": {
+ "hf_path": "voices/henri.safetensors",
+ "sha256": "2be79463e8a2cfe339a7d6088e397c22be4430c6796e76362a0f33f351bf940a"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.",
+ "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain",
+ "audio": "audio/henri.opus",
+ "sha256": "9a63ac8258c23aa750c871b4cc050da5658cd44b77448aad3cf1caef5ecb047b"
+ }
+ },
+ {
+ "name": "colette",
+ "language": "French",
+ "language_id": "fr",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "1406",
+ "source": {
+ "name": "Kyutai tts-voices, cml-tts french enhanced references",
+ "url": "https://huggingface.co/kyutai/tts-voices",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "colette.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/colette.opus",
+ "sha256": "324d657ffdc1d1651e9da4ad57258b1c54ee429b437f909ac29130e78cee21a9",
+ "duration_s": 7.24,
+ "construction": "pre-cut enhanced 10 s reference"
+ },
+ "profile": {
+ "hf_path": "voices/colette.safetensors",
+ "sha256": "bdccafdc54f700cb456e3f531d0f9da3e5cc0c9fcece8284db44773f99c162c9"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.",
+ "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain",
+ "audio": "audio/colette.opus",
+ "sha256": "e9f0084bb45e1b7647eec082dc03f43a110e10f915a2baf99d5776201ec664bd"
+ }
+ },
+ {
+ "name": "pim",
+ "language": "Dutch",
+ "language_id": "nl",
+ "gender": "M",
+ "donor": "pim (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "pim.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/pim.opus",
+ "sha256": "b7428579a45f688ebf6602fd94e6a92dfe34b20cf34e7aae3c73fb73b2182d29",
+ "duration_s": 11.0,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/pim.safetensors",
+ "sha256": "05c5b3bd921240acc39cc8ffd576ab71dcbc36e6238ce468292ae74c61190130"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.",
+ "work": "De pruimeboom (Hieronymus van Alphen) — public domain",
+ "audio": "audio/pim.opus",
+ "sha256": "c00535642d18623f4d2085fa0bde17bdd08c7c8b9b38ade8a85dfee7b9974b37"
+ }
+ },
+ {
+ "name": "nathalie",
+ "language": "Dutch",
+ "language_id": "nl",
+ "gender": "F",
+ "donor": "nathalie (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "nathalie.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/nathalie.opus",
+ "sha256": "bc09a866b18264d555a35522e3319ff820914f59a8c3dd8687c01ee16203e5fd",
+ "duration_s": 10.8,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/nathalie.safetensors",
+ "sha256": "a61158533e125600613719617ccc89ea4b66fbc9fbe149ee02e3b5c4113cc11d"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.",
+ "work": "De pruimeboom (Hieronymus van Alphen) — public domain",
+ "audio": "audio/nathalie.opus",
+ "sha256": "d2bf4970c1cf1d8f4437ee183e67ed5d1f67683249d26e36c6bf7cf3904490da"
+ }
+ },
+ {
+ "name": "dave",
+ "language": "Spanish",
+ "language_id": "es",
+ "gender": "M",
+ "donor": "dave (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "dave.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/dave.opus",
+ "sha256": "9e30010eafeb37abb20a9976e17bfa3d313f779d5703bba9a536791916b162d6",
+ "duration_s": 11.0,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/dave.safetensors",
+ "sha256": "14b5e58a6e5a457b516f5a9ae02d3a739449a02a92d0791b356377b68c366cb4"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.",
+ "work": "La mona (Felix Maria de Samaniego) — public domain",
+ "audio": "audio/dave.opus",
+ "sha256": "c1240b63c1f83b5279dccd028c18410d4051f590e695199a91bdd87af4c08d22"
+ }
+ },
+ {
+ "name": "carmen",
+ "language": "Spanish",
+ "language_id": "es",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "2308",
+ "source": {
+ "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)",
+ "url": "https://huggingface.co/datasets/ylacombe/cml-tts",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "carmen.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/carmen.opus",
+ "sha256": "ec9a38e7c13420543a004c4db665af8fb1b66a952f18edeb98d0099a14f186e3",
+ "duration_s": 14.37,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/carmen.safetensors",
+ "sha256": "c9387f3bc58e8e1dd41700339e886a546200597fe629e523282ba4dbd4ddc069"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 3,
+ "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.",
+ "work": "La mona (Felix Maria de Samaniego) — public domain",
+ "audio": "audio/carmen.opus",
+ "sha256": "aefb626d30e906f729e8ea4c7a178f3b0b3092851fdbb1af2578fb840ae2078c"
+ }
+ },
+ {
+ "name": "dante",
+ "language": "Italian",
+ "language_id": "it",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "12598",
+ "source": {
+ "name": "Multilingual LibriSpeech (facebook/multilingual_librispeech)",
+ "url": "https://huggingface.co/datasets/facebook/multilingual_librispeech",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "dante.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/dante.opus",
+ "sha256": "df142e760a870aaf53d71e945fcafbdfb2c5bcf14676c4c8b90ded229b9a48fe",
+ "duration_s": 14.0,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/dante.safetensors",
+ "sha256": "d0ba2cf9871b7f933ff5f0dbdca9578077599c2e8209fc6246727edd79c15924"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.",
+ "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain",
+ "audio": "audio/dante.opus",
+ "sha256": "9b05b0d44b83e208ea4c97b30a4b2e03cb323e5108adf0847f1572f18639fc04"
+ }
+ },
+ {
+ "name": "paola",
+ "language": "Italian",
+ "language_id": "it",
+ "gender": "F",
+ "donor": "paola (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "paola.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/paola.opus",
+ "sha256": "424e0cecda21fa34c5a05fb5f6bc029949f0485ac91e7fd45afc7645ff4cdbb9",
+ "duration_s": 11.0,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/paola.safetensors",
+ "sha256": "6c8bcd3d39dce92955833b1b9b2d483d1f8e48ac48fb772acbb63d45e4e00f6c"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.",
+ "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain",
+ "audio": "audio/paola.opus",
+ "sha256": "452ae96e4a6195a781c1aa521d7a856300ae814541856702824ad413527dabd3"
+ }
+ },
+ {
+ "name": "tugao",
+ "language": "Portuguese (European)",
+ "language_id": "pt",
+ "gender": "M",
+ "donor": "tugao (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "tugao.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/tugao.opus",
+ "sha256": "d6b40a71decdc9e9a560442f0c0e78f412b4d94365c9bddcb3e215ca4f452118",
+ "duration_s": 9.78,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/tugao.safetensors",
+ "sha256": "adcebf6fe42125bdc0aaca6ba3d23d79a30c8ad6a52aa65fd7970bed540eec43"
+ },
+ "speaker_similarity": 0.916,
+ "sample": {
+ "seed": 7,
+ "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.",
+ "work": "A Cidade e as Serras (Eca de Queiros) — public domain",
+ "audio": "audio/tugao.opus",
+ "sha256": "ddf3be70b986247fc879f90d5dc3f9e41e28d763e08ca10de1a44ea08e3f112b"
+ }
+ },
+ {
+ "name": "ines",
+ "language": "Portuguese (European)",
+ "language_id": "pt",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "7925",
+ "source": {
+ "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)",
+ "url": "https://huggingface.co/datasets/ylacombe/cml-tts",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "ines.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/ines.opus",
+ "sha256": "59f2fde8e98c8183e995e2576128ac4bbb186c08f677c1ac6daf2cd359bba7f4",
+ "duration_s": 10.44,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/ines.safetensors",
+ "sha256": "b7f76b52170dfcdcef0e8835956967bf2fbc4973883b6e99d6b2be7fd4f3a496"
+ },
+ "speaker_similarity": 0.96,
+ "sample": {
+ "seed": 7,
+ "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.",
+ "work": "A Cidade e as Serras (Eca de Queiros) — public domain",
+ "audio": "audio/ines.opus",
+ "sha256": "55c9cb8a3716ccfce280c25b577e344363fe3978b35577ce8c20ee6d4f3869ea"
+ },
+ "notes": "CML-TTS is LibriVox-derived; accent is Brazilian Portuguese, unlike tugao (European)."
+ },
+ {
+ "name": "nils",
+ "language": "Swedish",
+ "language_id": "sv",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": null,
+ "source": {
+ "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)",
+ "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "nils.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/nils.opus",
+ "sha256": "26680c82a976fe2d6f3202a97b50499274f1f0c36d329cbe6cdd6c3ebac1f0f0",
+ "duration_s": 11.0,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/nils.safetensors",
+ "sha256": "be75721c99547ab3ecdf1f8629101e9fa3ea1d569fc6e1220b8e2f259fb49f4e"
+ },
+ "speaker_similarity": 0.92,
+ "sample": {
+ "seed": 7,
+ "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.",
+ "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain",
+ "audio": "audio/nils.opus",
+ "sha256": "61b704118c82878a218d0db2b0fd3da20a70d0d442e2a36cc064ccc1a7f0679b"
+ }
+ },
+ {
+ "name": "selma",
+ "language": "Swedish",
+ "language_id": "sv",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": null,
+ "source": {
+ "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)",
+ "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "selma.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/selma.opus",
+ "sha256": "775d4db5a95ac9fea95b1f9b8d6a5b1917a3aefd466e153962ad195d83ed51d0",
+ "duration_s": 8.53,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/selma.safetensors",
+ "sha256": "d3ba9ab18c676db23028c9561ad79a183b35adae377d20d46f1d8f253ce6eeec"
+ },
+ "speaker_similarity": 0.925,
+ "sample": {
+ "seed": 7,
+ "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.",
+ "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain",
+ "audio": "audio/selma.opus",
+ "sha256": "70f186995d8b6aac965662cdff346ec5211bef98276c7ef94a69c096dccd0782"
+ }
+ },
+ {
+ "name": "soren",
+ "language": "Danish",
+ "language_id": "da",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "37",
+ "source": {
+ "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)",
+ "url": "https://huggingface.co/datasets/alexandrainst/nst-da",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "soren.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/soren.opus",
+ "sha256": "a0d7510fa2de66b8bf656c5f2ff93ddf462878ce31799860ca242bcad64487fb",
+ "duration_s": 7.38,
+ "construction": "single continuous clip, close mic"
+ },
+ "profile": {
+ "hf_path": "voices/soren.safetensors",
+ "sha256": "30d9fe9ee02928cd65ef7f4cb286c401c1809f07c13c19de79ffa375a027412e"
+ },
+ "speaker_similarity": 0.893,
+ "sample": {
+ "seed": 7,
+ "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.",
+ "work": "Den grimme aelling (H. C. Andersen) — public domain",
+ "audio": "audio/soren.opus",
+ "sha256": "f9841dd5162d30df3e67aee92128bd3708f4eff6c3d1c5c177091765cf04af2c"
+ }
+ },
+ {
+ "name": "freja",
+ "language": "Danish",
+ "language_id": "da",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "35",
+ "source": {
+ "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)",
+ "url": "https://huggingface.co/datasets/alexandrainst/nst-da",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "freja.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/freja.opus",
+ "sha256": "078baf737c9acc6f567dd87a8b3eca897becb0e83d9209279bf8314830d2678a",
+ "duration_s": 8.78,
+ "construction": "single continuous clip, close mic"
+ },
+ "profile": {
+ "hf_path": "voices/freja.safetensors",
+ "sha256": "0a3f913b9b47c529393f843850228bd8eee6a883b4c138978fd3afa7758561e8"
+ },
+ "speaker_similarity": 0.907,
+ "sample": {
+ "seed": 7,
+ "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.",
+ "work": "Den grimme aelling (H. C. Andersen) — public domain",
+ "audio": "audio/freja.opus",
+ "sha256": "cd785cc844250a5ae97010e2a3b0ee9260c5a24d016350d931aa7ec776958812"
+ }
+ }
+]
diff --git a/hf-loudkit/requirements.txt b/hf-loudkit/requirements.txt
new file mode 100644
index 0000000000000000000000000000000000000000..60b4bdb741c3b04aa1bb164e82cbfb4bbb071a24
--- /dev/null
+++ b/hf-loudkit/requirements.txt
@@ -0,0 +1,18 @@
+# The library under demo. 0.1.0 is on PyPI, so this installs from the registry
+# rather than carrying a wheel beside the app.
+#
+# torch the cuda backend this Space runs on
+# audio soundfile, for Result.save
+# enroll torchaudio + librosa, for the Clone tab
+# hub huggingface_hub, to resolve loudreader/loudr-1
+loudkit[torch,audio,enroll,hub]==0.1.0
+
+# loudkit floors torch at >=2.4 and sets no ceiling, and ZeroGPU runs a fixed
+# set of builds: 2.8.0, 2.9.1, 2.10.0, 2.11.0. Pinned here so the resolver
+# cannot land outside that window. torchaudio tracks torch version for version.
+torch==2.8.0
+torchaudio==2.8.0
+
+# The ZeroGPU decorator. Present on ZeroGPU hardware already; named here so a
+# local run or a duplicated Space installs it too.
+spaces>=0.42
diff --git a/hf-loudkit/voices.json b/hf-loudkit/voices.json
new file mode 100644
index 0000000000000000000000000000000000000000..ac89f1f9159d32cd3c35686f32448d57a27284b2
--- /dev/null
+++ b/hf-loudkit/voices.json
@@ -0,0 +1,683 @@
+[
+ {
+ "name": "darkman",
+ "language": "Polish",
+ "language_id": "pl",
+ "gender": "M",
+ "donor": "darkman (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "darkman.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/darkman.opus",
+ "sha256": "d228db9d17b1ddb4ba450f16a6b8650bb0a8550521b78a914cb9e573cd24d576",
+ "duration_s": 9.78,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/darkman.safetensors",
+ "sha256": "78988a6769f90c209240f9f285eb9cdd760d1ba471cfa7d6e67db632599cfa1b"
+ },
+ "speaker_similarity": 0.935,
+ "sample": {
+ "seed": 7,
+ "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.",
+ "work": "Quo Vadis (Henryk Sienkiewicz) — public domain",
+ "audio": "audio/darkman.opus",
+ "sha256": "b30f98cf0452e9c694fcbafc27b413041c87022238f5424fc1aa452c8eb156a5"
+ }
+ },
+ {
+ "name": "gosia",
+ "language": "Polish",
+ "language_id": "pl",
+ "gender": "F",
+ "donor": "gosia (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "gosia.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/gosia.opus",
+ "sha256": "02c1d09db8b097c66dd8705747a6ef138596f5f91c3213b1cd430ae75ca2279a",
+ "duration_s": 10.86,
+ "construction": "concatenated long donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/gosia.safetensors",
+ "sha256": "e115cf64bfa0ad15770a50c0be8097312a58d41f84c7bb4f01ecbdbac5d21823"
+ },
+ "speaker_similarity": 0.917,
+ "sample": {
+ "seed": 7,
+ "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.",
+ "work": "Quo Vadis (Henryk Sienkiewicz) — public domain",
+ "audio": "audio/gosia.opus",
+ "sha256": "4ba1107d7f807ac2e0941167d1bc4a7cfb2039a4b0482a3bc798db5eeefe17e5"
+ }
+ },
+ {
+ "name": "joe",
+ "language": "English",
+ "language_id": "en",
+ "gender": "M",
+ "donor": "joe (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "joe.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/joe.opus",
+ "sha256": "591c8d6a5799a4266d029762dc0f671ac9f0455382aeef1e763ca206d6fdd164",
+ "duration_s": 10.99,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/joe.safetensors",
+ "sha256": "cfc3687fdb1f6b56fe13e252ca4a5bd681cef0702dc5cd56d4ebbc60e1aea6e8"
+ },
+ "speaker_similarity": 0.924,
+ "sample": {
+ "seed": 11,
+ "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.",
+ "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain",
+ "audio": "audio/joe.opus",
+ "sha256": "83e0bf0a1f6b03e047de66871536c3ff77291041d0a3f2907b6990e580519fea"
+ }
+ },
+ {
+ "name": "kathleen",
+ "language": "English",
+ "language_id": "en",
+ "gender": "F",
+ "donor": "kathleen (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "kathleen.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/kathleen.opus",
+ "sha256": "f43a24359a333e1f723c5f10d7dc799ec5880abf2b5b9321a7d5f8c7064d6f73",
+ "duration_s": 11.0,
+ "construction": "concatenated donation clips; prompt texts from the CMU ARCTIC prompt list (public-domain Gutenberg novels), recording is kathleen's own CC0 donation"
+ },
+ "profile": {
+ "hf_path": "voices/kathleen.safetensors",
+ "sha256": "6a845e4eb11989fc9bd8e19eed7c5ac585e6d08e1ec824211f0eb240ce523d29"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.",
+ "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain",
+ "audio": "audio/kathleen.opus",
+ "sha256": "48615517c1d1d2d55dbb7f74f37982918236e33f079947459d56c116cd70b7a3"
+ }
+ },
+ {
+ "name": "thorsten",
+ "language": "German",
+ "language_id": "de",
+ "gender": "M",
+ "donor": "Thorsten Mueller",
+ "speaker_id": null,
+ "source": {
+ "name": "Thorsten-Voice TV-44kHz-Full",
+ "url": "https://huggingface.co/datasets/Thorsten-Voice/TV-44kHz-Full",
+ "license": "CC0-1.0",
+ "consent": "voice deliberately donated by Thorsten Mueller for TTS"
+ },
+ "reference": {
+ "source_filename": "thorsten.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/thorsten.opus",
+ "sha256": "37805d1136f727f7530e714b772c23224ae1c6f6931ea5b78399cf286694de3e",
+ "duration_s": 6.37,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/thorsten.safetensors",
+ "sha256": "87873fec8d3dceaa643ab1ab07cf5143364e659dd6b3caec707f51408ac2154a"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.",
+ "work": "Max und Moritz (Wilhelm Busch) — public domain",
+ "audio": "audio/thorsten.opus",
+ "sha256": "d45938440534f7fa11afe388fa41a9a56879809a65c8477605220ed53d5a938b"
+ }
+ },
+ {
+ "name": "kerstin",
+ "language": "German",
+ "language_id": "de",
+ "gender": "F",
+ "donor": "kerstin (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "kerstin.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/kerstin.opus",
+ "sha256": "34d7c0db776bcda98e56bc130ced840debe0b12a73f4b03c556b30fc356db187",
+ "duration_s": 9.79,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/kerstin.safetensors",
+ "sha256": "d096ecde83dfc5de7a927acf67a5304c55f863ca722f826c67fafd31c48609a6"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.",
+ "work": "Max und Moritz (Wilhelm Busch) — public domain",
+ "audio": "audio/kerstin.opus",
+ "sha256": "dc1990efd85eafee75810f45fd259505d4840579f008aba9f38ca1a7c42e839b"
+ }
+ },
+ {
+ "name": "henri",
+ "language": "French",
+ "language_id": "fr",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "10087",
+ "source": {
+ "name": "Kyutai tts-voices, cml-tts french enhanced references",
+ "url": "https://huggingface.co/kyutai/tts-voices",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "henri.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/henri.opus",
+ "sha256": "9185f21daf7bcbb9a53ea06b059438ea83b7315ef4f852d69d56afc562061fc0",
+ "duration_s": 6.56,
+ "construction": "pre-cut enhanced 10 s reference"
+ },
+ "profile": {
+ "hf_path": "voices/henri.safetensors",
+ "sha256": "2be79463e8a2cfe339a7d6088e397c22be4430c6796e76362a0f33f351bf940a"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.",
+ "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain",
+ "audio": "audio/henri.opus",
+ "sha256": "9a63ac8258c23aa750c871b4cc050da5658cd44b77448aad3cf1caef5ecb047b"
+ }
+ },
+ {
+ "name": "colette",
+ "language": "French",
+ "language_id": "fr",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "1406",
+ "source": {
+ "name": "Kyutai tts-voices, cml-tts french enhanced references",
+ "url": "https://huggingface.co/kyutai/tts-voices",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "colette.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/colette.opus",
+ "sha256": "324d657ffdc1d1651e9da4ad57258b1c54ee429b437f909ac29130e78cee21a9",
+ "duration_s": 7.24,
+ "construction": "pre-cut enhanced 10 s reference"
+ },
+ "profile": {
+ "hf_path": "voices/colette.safetensors",
+ "sha256": "bdccafdc54f700cb456e3f531d0f9da3e5cc0c9fcece8284db44773f99c162c9"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.",
+ "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain",
+ "audio": "audio/colette.opus",
+ "sha256": "e9f0084bb45e1b7647eec082dc03f43a110e10f915a2baf99d5776201ec664bd"
+ }
+ },
+ {
+ "name": "pim",
+ "language": "Dutch",
+ "language_id": "nl",
+ "gender": "M",
+ "donor": "pim (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "pim.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/pim.opus",
+ "sha256": "b7428579a45f688ebf6602fd94e6a92dfe34b20cf34e7aae3c73fb73b2182d29",
+ "duration_s": 11.0,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/pim.safetensors",
+ "sha256": "05c5b3bd921240acc39cc8ffd576ab71dcbc36e6238ce468292ae74c61190130"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.",
+ "work": "De pruimeboom (Hieronymus van Alphen) — public domain",
+ "audio": "audio/pim.opus",
+ "sha256": "c00535642d18623f4d2085fa0bde17bdd08c7c8b9b38ade8a85dfee7b9974b37"
+ }
+ },
+ {
+ "name": "nathalie",
+ "language": "Dutch",
+ "language_id": "nl",
+ "gender": "F",
+ "donor": "nathalie (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "nathalie.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/nathalie.opus",
+ "sha256": "bc09a866b18264d555a35522e3319ff820914f59a8c3dd8687c01ee16203e5fd",
+ "duration_s": 10.8,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/nathalie.safetensors",
+ "sha256": "a61158533e125600613719617ccc89ea4b66fbc9fbe149ee02e3b5c4113cc11d"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.",
+ "work": "De pruimeboom (Hieronymus van Alphen) — public domain",
+ "audio": "audio/nathalie.opus",
+ "sha256": "d2bf4970c1cf1d8f4437ee183e67ed5d1f67683249d26e36c6bf7cf3904490da"
+ }
+ },
+ {
+ "name": "dave",
+ "language": "Spanish",
+ "language_id": "es",
+ "gender": "M",
+ "donor": "dave (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "dave.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/dave.opus",
+ "sha256": "9e30010eafeb37abb20a9976e17bfa3d313f779d5703bba9a536791916b162d6",
+ "duration_s": 11.0,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/dave.safetensors",
+ "sha256": "14b5e58a6e5a457b516f5a9ae02d3a739449a02a92d0791b356377b68c366cb4"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.",
+ "work": "La mona (Felix Maria de Samaniego) — public domain",
+ "audio": "audio/dave.opus",
+ "sha256": "c1240b63c1f83b5279dccd028c18410d4051f590e695199a91bdd87af4c08d22"
+ }
+ },
+ {
+ "name": "carmen",
+ "language": "Spanish",
+ "language_id": "es",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "2308",
+ "source": {
+ "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)",
+ "url": "https://huggingface.co/datasets/ylacombe/cml-tts",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "carmen.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/carmen.opus",
+ "sha256": "ec9a38e7c13420543a004c4db665af8fb1b66a952f18edeb98d0099a14f186e3",
+ "duration_s": 14.37,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/carmen.safetensors",
+ "sha256": "c9387f3bc58e8e1dd41700339e886a546200597fe629e523282ba4dbd4ddc069"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 3,
+ "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.",
+ "work": "La mona (Felix Maria de Samaniego) — public domain",
+ "audio": "audio/carmen.opus",
+ "sha256": "aefb626d30e906f729e8ea4c7a178f3b0b3092851fdbb1af2578fb840ae2078c"
+ }
+ },
+ {
+ "name": "dante",
+ "language": "Italian",
+ "language_id": "it",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "12598",
+ "source": {
+ "name": "Multilingual LibriSpeech (facebook/multilingual_librispeech)",
+ "url": "https://huggingface.co/datasets/facebook/multilingual_librispeech",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "dante.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/dante.opus",
+ "sha256": "df142e760a870aaf53d71e945fcafbdfb2c5bcf14676c4c8b90ded229b9a48fe",
+ "duration_s": 14.0,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/dante.safetensors",
+ "sha256": "d0ba2cf9871b7f933ff5f0dbdca9578077599c2e8209fc6246727edd79c15924"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.",
+ "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain",
+ "audio": "audio/dante.opus",
+ "sha256": "9b05b0d44b83e208ea4c97b30a4b2e03cb323e5108adf0847f1572f18639fc04"
+ }
+ },
+ {
+ "name": "paola",
+ "language": "Italian",
+ "language_id": "it",
+ "gender": "F",
+ "donor": "paola (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "paola.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/paola.opus",
+ "sha256": "424e0cecda21fa34c5a05fb5f6bc029949f0485ac91e7fd45afc7645ff4cdbb9",
+ "duration_s": 11.0,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/paola.safetensors",
+ "sha256": "6c8bcd3d39dce92955833b1b9b2d483d1f8e48ac48fb772acbb63d45e4e00f6c"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.",
+ "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain",
+ "audio": "audio/paola.opus",
+ "sha256": "452ae96e4a6195a781c1aa521d7a856300ae814541856702824ad413527dabd3"
+ }
+ },
+ {
+ "name": "tugao",
+ "language": "Portuguese (European)",
+ "language_id": "pt",
+ "gender": "M",
+ "donor": "tugao (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "tugao.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/tugao.opus",
+ "sha256": "d6b40a71decdc9e9a560442f0c0e78f412b4d94365c9bddcb3e215ca4f452118",
+ "duration_s": 9.78,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/tugao.safetensors",
+ "sha256": "adcebf6fe42125bdc0aaca6ba3d23d79a30c8ad6a52aa65fd7970bed540eec43"
+ },
+ "speaker_similarity": 0.916,
+ "sample": {
+ "seed": 7,
+ "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.",
+ "work": "A Cidade e as Serras (Eca de Queiros) — public domain",
+ "audio": "audio/tugao.opus",
+ "sha256": "ddf3be70b986247fc879f90d5dc3f9e41e28d763e08ca10de1a44ea08e3f112b"
+ }
+ },
+ {
+ "name": "ines",
+ "language": "Portuguese (European)",
+ "language_id": "pt",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "7925",
+ "source": {
+ "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)",
+ "url": "https://huggingface.co/datasets/ylacombe/cml-tts",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "ines.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/ines.opus",
+ "sha256": "59f2fde8e98c8183e995e2576128ac4bbb186c08f677c1ac6daf2cd359bba7f4",
+ "duration_s": 10.44,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/ines.safetensors",
+ "sha256": "b7f76b52170dfcdcef0e8835956967bf2fbc4973883b6e99d6b2be7fd4f3a496"
+ },
+ "speaker_similarity": 0.96,
+ "sample": {
+ "seed": 7,
+ "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.",
+ "work": "A Cidade e as Serras (Eca de Queiros) — public domain",
+ "audio": "audio/ines.opus",
+ "sha256": "55c9cb8a3716ccfce280c25b577e344363fe3978b35577ce8c20ee6d4f3869ea"
+ },
+ "notes": "CML-TTS is LibriVox-derived; accent is Brazilian Portuguese, unlike tugao (European)."
+ },
+ {
+ "name": "nils",
+ "language": "Swedish",
+ "language_id": "sv",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": null,
+ "source": {
+ "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)",
+ "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "nils.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/nils.opus",
+ "sha256": "26680c82a976fe2d6f3202a97b50499274f1f0c36d329cbe6cdd6c3ebac1f0f0",
+ "duration_s": 11.0,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/nils.safetensors",
+ "sha256": "be75721c99547ab3ecdf1f8629101e9fa3ea1d569fc6e1220b8e2f259fb49f4e"
+ },
+ "speaker_similarity": 0.92,
+ "sample": {
+ "seed": 7,
+ "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.",
+ "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain",
+ "audio": "audio/nils.opus",
+ "sha256": "61b704118c82878a218d0db2b0fd3da20a70d0d442e2a36cc064ccc1a7f0679b"
+ }
+ },
+ {
+ "name": "selma",
+ "language": "Swedish",
+ "language_id": "sv",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": null,
+ "source": {
+ "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)",
+ "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "selma.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/selma.opus",
+ "sha256": "775d4db5a95ac9fea95b1f9b8d6a5b1917a3aefd466e153962ad195d83ed51d0",
+ "duration_s": 8.53,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/selma.safetensors",
+ "sha256": "d3ba9ab18c676db23028c9561ad79a183b35adae377d20d46f1d8f253ce6eeec"
+ },
+ "speaker_similarity": 0.925,
+ "sample": {
+ "seed": 7,
+ "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.",
+ "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain",
+ "audio": "audio/selma.opus",
+ "sha256": "70f186995d8b6aac965662cdff346ec5211bef98276c7ef94a69c096dccd0782"
+ }
+ },
+ {
+ "name": "soren",
+ "language": "Danish",
+ "language_id": "da",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "37",
+ "source": {
+ "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)",
+ "url": "https://huggingface.co/datasets/alexandrainst/nst-da",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "soren.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/soren.opus",
+ "sha256": "a0d7510fa2de66b8bf656c5f2ff93ddf462878ce31799860ca242bcad64487fb",
+ "duration_s": 7.38,
+ "construction": "single continuous clip, close mic"
+ },
+ "profile": {
+ "hf_path": "voices/soren.safetensors",
+ "sha256": "30d9fe9ee02928cd65ef7f4cb286c401c1809f07c13c19de79ffa375a027412e"
+ },
+ "speaker_similarity": 0.893,
+ "sample": {
+ "seed": 7,
+ "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.",
+ "work": "Den grimme aelling (H. C. Andersen) — public domain",
+ "audio": "audio/soren.opus",
+ "sha256": "f9841dd5162d30df3e67aee92128bd3708f4eff6c3d1c5c177091765cf04af2c"
+ }
+ },
+ {
+ "name": "freja",
+ "language": "Danish",
+ "language_id": "da",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "35",
+ "source": {
+ "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)",
+ "url": "https://huggingface.co/datasets/alexandrainst/nst-da",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "freja.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/freja.opus",
+ "sha256": "078baf737c9acc6f567dd87a8b3eca897becb0e83d9209279bf8314830d2678a",
+ "duration_s": 8.78,
+ "construction": "single continuous clip, close mic"
+ },
+ "profile": {
+ "hf_path": "voices/freja.safetensors",
+ "sha256": "0a3f913b9b47c529393f843850228bd8eee6a883b4c138978fd3afa7758561e8"
+ },
+ "speaker_similarity": 0.907,
+ "sample": {
+ "seed": 7,
+ "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.",
+ "work": "Den grimme aelling (H. C. Andersen) — public domain",
+ "audio": "audio/freja.opus",
+ "sha256": "cd785cc844250a5ae97010e2a3b0ee9260c5a24d016350d931aa7ec776958812"
+ }
+ }
+]
diff --git a/requirements.txt b/requirements.txt
new file mode 100644
index 0000000000000000000000000000000000000000..60b4bdb741c3b04aa1bb164e82cbfb4bbb071a24
--- /dev/null
+++ b/requirements.txt
@@ -0,0 +1,18 @@
+# The library under demo. 0.1.0 is on PyPI, so this installs from the registry
+# rather than carrying a wheel beside the app.
+#
+# torch the cuda backend this Space runs on
+# audio soundfile, for Result.save
+# enroll torchaudio + librosa, for the Clone tab
+# hub huggingface_hub, to resolve loudreader/loudr-1
+loudkit[torch,audio,enroll,hub]==0.1.0
+
+# loudkit floors torch at >=2.4 and sets no ceiling, and ZeroGPU runs a fixed
+# set of builds: 2.8.0, 2.9.1, 2.10.0, 2.11.0. Pinned here so the resolver
+# cannot land outside that window. torchaudio tracks torch version for version.
+torch==2.8.0
+torchaudio==2.8.0
+
+# The ZeroGPU decorator. Present on ZeroGPU hardware already; named here so a
+# local run or a duplicated Space installs it too.
+spaces>=0.42
diff --git a/voices.json b/voices.json
new file mode 100644
index 0000000000000000000000000000000000000000..ac89f1f9159d32cd3c35686f32448d57a27284b2
--- /dev/null
+++ b/voices.json
@@ -0,0 +1,683 @@
+[
+ {
+ "name": "darkman",
+ "language": "Polish",
+ "language_id": "pl",
+ "gender": "M",
+ "donor": "darkman (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "darkman.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/darkman.opus",
+ "sha256": "d228db9d17b1ddb4ba450f16a6b8650bb0a8550521b78a914cb9e573cd24d576",
+ "duration_s": 9.78,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/darkman.safetensors",
+ "sha256": "78988a6769f90c209240f9f285eb9cdd760d1ba471cfa7d6e67db632599cfa1b"
+ },
+ "speaker_similarity": 0.935,
+ "sample": {
+ "seed": 7,
+ "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.",
+ "work": "Quo Vadis (Henryk Sienkiewicz) — public domain",
+ "audio": "audio/darkman.opus",
+ "sha256": "b30f98cf0452e9c694fcbafc27b413041c87022238f5424fc1aa452c8eb156a5"
+ }
+ },
+ {
+ "name": "gosia",
+ "language": "Polish",
+ "language_id": "pl",
+ "gender": "F",
+ "donor": "gosia (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "gosia.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/gosia.opus",
+ "sha256": "02c1d09db8b097c66dd8705747a6ef138596f5f91c3213b1cd430ae75ca2279a",
+ "duration_s": 10.86,
+ "construction": "concatenated long donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/gosia.safetensors",
+ "sha256": "e115cf64bfa0ad15770a50c0be8097312a58d41f84c7bb4f01ecbdbac5d21823"
+ },
+ "speaker_similarity": 0.917,
+ "sample": {
+ "seed": 7,
+ "text": "Petroniusz obudził się zaledwie koło południa, a jak zwykle, zmęczony bardzo.",
+ "work": "Quo Vadis (Henryk Sienkiewicz) — public domain",
+ "audio": "audio/gosia.opus",
+ "sha256": "4ba1107d7f807ac2e0941167d1bc4a7cfb2039a4b0482a3bc798db5eeefe17e5"
+ }
+ },
+ {
+ "name": "joe",
+ "language": "English",
+ "language_id": "en",
+ "gender": "M",
+ "donor": "joe (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "joe.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/joe.opus",
+ "sha256": "591c8d6a5799a4266d029762dc0f671ac9f0455382aeef1e763ca206d6fdd164",
+ "duration_s": 10.99,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/joe.safetensors",
+ "sha256": "cfc3687fdb1f6b56fe13e252ca4a5bd681cef0702dc5cd56d4ebbc60e1aea6e8"
+ },
+ "speaker_similarity": 0.924,
+ "sample": {
+ "seed": 11,
+ "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.",
+ "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain",
+ "audio": "audio/joe.opus",
+ "sha256": "83e0bf0a1f6b03e047de66871536c3ff77291041d0a3f2907b6990e580519fea"
+ }
+ },
+ {
+ "name": "kathleen",
+ "language": "English",
+ "language_id": "en",
+ "gender": "F",
+ "donor": "kathleen (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "kathleen.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/kathleen.opus",
+ "sha256": "f43a24359a333e1f723c5f10d7dc799ec5880abf2b5b9321a7d5f8c7064d6f73",
+ "duration_s": 11.0,
+ "construction": "concatenated donation clips; prompt texts from the CMU ARCTIC prompt list (public-domain Gutenberg novels), recording is kathleen's own CC0 donation"
+ },
+ "profile": {
+ "hf_path": "voices/kathleen.safetensors",
+ "sha256": "6a845e4eb11989fc9bd8e19eed7c5ac585e6d08e1ec824211f0eb240ce523d29"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Alice was beginning to get very tired of sitting by her sister on the bank, and of having nothing to do: once or twice she had peeped into the book her sister was reading.",
+ "work": "Alice's Adventures in Wonderland (Lewis Carroll) — public domain",
+ "audio": "audio/kathleen.opus",
+ "sha256": "48615517c1d1d2d55dbb7f74f37982918236e33f079947459d56c116cd70b7a3"
+ }
+ },
+ {
+ "name": "thorsten",
+ "language": "German",
+ "language_id": "de",
+ "gender": "M",
+ "donor": "Thorsten Mueller",
+ "speaker_id": null,
+ "source": {
+ "name": "Thorsten-Voice TV-44kHz-Full",
+ "url": "https://huggingface.co/datasets/Thorsten-Voice/TV-44kHz-Full",
+ "license": "CC0-1.0",
+ "consent": "voice deliberately donated by Thorsten Mueller for TTS"
+ },
+ "reference": {
+ "source_filename": "thorsten.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/thorsten.opus",
+ "sha256": "37805d1136f727f7530e714b772c23224ae1c6f6931ea5b78399cf286694de3e",
+ "duration_s": 6.37,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/thorsten.safetensors",
+ "sha256": "87873fec8d3dceaa643ab1ab07cf5143364e659dd6b3caec707f51408ac2154a"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.",
+ "work": "Max und Moritz (Wilhelm Busch) — public domain",
+ "audio": "audio/thorsten.opus",
+ "sha256": "d45938440534f7fa11afe388fa41a9a56879809a65c8477605220ed53d5a938b"
+ }
+ },
+ {
+ "name": "kerstin",
+ "language": "German",
+ "language_id": "de",
+ "gender": "F",
+ "donor": "kerstin (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "kerstin.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/kerstin.opus",
+ "sha256": "34d7c0db776bcda98e56bc130ced840debe0b12a73f4b03c556b30fc356db187",
+ "duration_s": 9.79,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/kerstin.safetensors",
+ "sha256": "d096ecde83dfc5de7a927acf67a5304c55f863ca722f826c67fafd31c48609a6"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Ach, was muss man oft von bösen Kindern hören oder lesen! Wie zum Beispiel hier von diesen, welche Max und Moritz hießen.",
+ "work": "Max und Moritz (Wilhelm Busch) — public domain",
+ "audio": "audio/kerstin.opus",
+ "sha256": "dc1990efd85eafee75810f45fd259505d4840579f008aba9f38ca1a7c42e839b"
+ }
+ },
+ {
+ "name": "henri",
+ "language": "French",
+ "language_id": "fr",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "10087",
+ "source": {
+ "name": "Kyutai tts-voices, cml-tts french enhanced references",
+ "url": "https://huggingface.co/kyutai/tts-voices",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "henri.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/henri.opus",
+ "sha256": "9185f21daf7bcbb9a53ea06b059438ea83b7315ef4f852d69d56afc562061fc0",
+ "duration_s": 6.56,
+ "construction": "pre-cut enhanced 10 s reference"
+ },
+ "profile": {
+ "hf_path": "voices/henri.safetensors",
+ "sha256": "2be79463e8a2cfe339a7d6088e397c22be4430c6796e76362a0f33f351bf940a"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.",
+ "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain",
+ "audio": "audio/henri.opus",
+ "sha256": "9a63ac8258c23aa750c871b4cc050da5658cd44b77448aad3cf1caef5ecb047b"
+ }
+ },
+ {
+ "name": "colette",
+ "language": "French",
+ "language_id": "fr",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "1406",
+ "source": {
+ "name": "Kyutai tts-voices, cml-tts french enhanced references",
+ "url": "https://huggingface.co/kyutai/tts-voices",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "colette.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/colette.opus",
+ "sha256": "324d657ffdc1d1651e9da4ad57258b1c54ee429b437f909ac29130e78cee21a9",
+ "duration_s": 7.24,
+ "construction": "pre-cut enhanced 10 s reference"
+ },
+ "profile": {
+ "hf_path": "voices/colette.safetensors",
+ "sha256": "bdccafdc54f700cb456e3f531d0f9da3e5cc0c9fcece8284db44773f99c162c9"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Maître Corbeau, sur un arbre perché, tenait en son bec un fromage. Maître Renard, par l'odeur alléché, lui tint à peu près ce langage: Hé! bonjour, Monsieur du Corbeau.",
+ "work": "Le Corbeau et le Renard (Jean de La Fontaine) — public domain",
+ "audio": "audio/colette.opus",
+ "sha256": "e9f0084bb45e1b7647eec082dc03f43a110e10f915a2baf99d5776201ec664bd"
+ }
+ },
+ {
+ "name": "pim",
+ "language": "Dutch",
+ "language_id": "nl",
+ "gender": "M",
+ "donor": "pim (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "pim.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/pim.opus",
+ "sha256": "b7428579a45f688ebf6602fd94e6a92dfe34b20cf34e7aae3c73fb73b2182d29",
+ "duration_s": 11.0,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/pim.safetensors",
+ "sha256": "05c5b3bd921240acc39cc8ffd576ab71dcbc36e6238ce468292ae74c61190130"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.",
+ "work": "De pruimeboom (Hieronymus van Alphen) — public domain",
+ "audio": "audio/pim.opus",
+ "sha256": "c00535642d18623f4d2085fa0bde17bdd08c7c8b9b38ade8a85dfee7b9974b37"
+ }
+ },
+ {
+ "name": "nathalie",
+ "language": "Dutch",
+ "language_id": "nl",
+ "gender": "F",
+ "donor": "nathalie (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "nathalie.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/nathalie.opus",
+ "sha256": "bc09a866b18264d555a35522e3319ff820914f59a8c3dd8687c01ee16203e5fd",
+ "duration_s": 10.8,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/nathalie.safetensors",
+ "sha256": "a61158533e125600613719617ccc89ea4b66fbc9fbe149ee02e3b5c4113cc11d"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Jantje zag eens pruimen hangen, o! als eieren zo groot. 't Scheen, dat Jantje wou gaan plukken, schoon zijn vader 't hem verbood.",
+ "work": "De pruimeboom (Hieronymus van Alphen) — public domain",
+ "audio": "audio/nathalie.opus",
+ "sha256": "d2bf4970c1cf1d8f4437ee183e67ed5d1f67683249d26e36c6bf7cf3904490da"
+ }
+ },
+ {
+ "name": "dave",
+ "language": "Spanish",
+ "language_id": "es",
+ "gender": "M",
+ "donor": "dave (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "dave.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/dave.opus",
+ "sha256": "9e30010eafeb37abb20a9976e17bfa3d313f779d5703bba9a536791916b162d6",
+ "duration_s": 11.0,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/dave.safetensors",
+ "sha256": "14b5e58a6e5a457b516f5a9ae02d3a739449a02a92d0791b356377b68c366cb4"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.",
+ "work": "La mona (Felix Maria de Samaniego) — public domain",
+ "audio": "audio/dave.opus",
+ "sha256": "c1240b63c1f83b5279dccd028c18410d4051f590e695199a91bdd87af4c08d22"
+ }
+ },
+ {
+ "name": "carmen",
+ "language": "Spanish",
+ "language_id": "es",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "2308",
+ "source": {
+ "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)",
+ "url": "https://huggingface.co/datasets/ylacombe/cml-tts",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "carmen.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/carmen.opus",
+ "sha256": "ec9a38e7c13420543a004c4db665af8fb1b66a952f18edeb98d0099a14f186e3",
+ "duration_s": 14.37,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/carmen.safetensors",
+ "sha256": "c9387f3bc58e8e1dd41700339e886a546200597fe629e523282ba4dbd4ddc069"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 3,
+ "text": "Subió una mona a un nogal, y cogiendo una nuez verde, en la cáscara la muerde, con que le supo muy mal.",
+ "work": "La mona (Felix Maria de Samaniego) — public domain",
+ "audio": "audio/carmen.opus",
+ "sha256": "aefb626d30e906f729e8ea4c7a178f3b0b3092851fdbb1af2578fb840ae2078c"
+ }
+ },
+ {
+ "name": "dante",
+ "language": "Italian",
+ "language_id": "it",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "12598",
+ "source": {
+ "name": "Multilingual LibriSpeech (facebook/multilingual_librispeech)",
+ "url": "https://huggingface.co/datasets/facebook/multilingual_librispeech",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "dante.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/dante.opus",
+ "sha256": "df142e760a870aaf53d71e945fcafbdfb2c5bcf14676c4c8b90ded229b9a48fe",
+ "duration_s": 14.0,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/dante.safetensors",
+ "sha256": "d0ba2cf9871b7f933ff5f0dbdca9578077599c2e8209fc6246727edd79c15924"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.",
+ "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain",
+ "audio": "audio/dante.opus",
+ "sha256": "9b05b0d44b83e208ea4c97b30a4b2e03cb323e5108adf0847f1572f18639fc04"
+ }
+ },
+ {
+ "name": "paola",
+ "language": "Italian",
+ "language_id": "it",
+ "gender": "F",
+ "donor": "paola (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "paola.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/paola.opus",
+ "sha256": "424e0cecda21fa34c5a05fb5f6bc029949f0485ac91e7fd45afc7645ff4cdbb9",
+ "duration_s": 11.0,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/paola.safetensors",
+ "sha256": "6c8bcd3d39dce92955833b1b9b2d483d1f8e48ac48fb772acbb63d45e4e00f6c"
+ },
+ "speaker_similarity": null,
+ "sample": {
+ "seed": 7,
+ "text": "C'era una volta... Un re! diranno subito i miei piccoli lettori. No, ragazzi, avete sbagliato: c'era una volta un pezzo di legno.",
+ "work": "Le avventure di Pinocchio (Carlo Collodi) — public domain",
+ "audio": "audio/paola.opus",
+ "sha256": "452ae96e4a6195a781c1aa521d7a856300ae814541856702824ad413527dabd3"
+ }
+ },
+ {
+ "name": "tugao",
+ "language": "Portuguese (European)",
+ "language_id": "pt",
+ "gender": "M",
+ "donor": "tugao (donor alias)",
+ "speaker_id": null,
+ "source": {
+ "name": "Home Assistant / Piper voice-datasets (voice donations recorded for TTS)",
+ "url": "https://github.com/NabuCasa/voice-datasets",
+ "license": "CC0-1.0",
+ "consent": "recorded and donated expressly for building TTS voices"
+ },
+ "reference": {
+ "source_filename": "tugao.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/tugao.opus",
+ "sha256": "d6b40a71decdc9e9a560442f0c0e78f412b4d94365c9bddcb3e215ca4f452118",
+ "duration_s": 9.78,
+ "construction": "concatenated donation clips, 120 ms gaps"
+ },
+ "profile": {
+ "hf_path": "voices/tugao.safetensors",
+ "sha256": "adcebf6fe42125bdc0aaca6ba3d23d79a30c8ad6a52aa65fd7970bed540eec43"
+ },
+ "speaker_similarity": 0.916,
+ "sample": {
+ "seed": 7,
+ "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.",
+ "work": "A Cidade e as Serras (Eca de Queiros) — public domain",
+ "audio": "audio/tugao.opus",
+ "sha256": "ddf3be70b986247fc879f90d5dc3f9e41e28d763e08ca10de1a44ea08e3f112b"
+ }
+ },
+ {
+ "name": "ines",
+ "language": "Portuguese (European)",
+ "language_id": "pt",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "7925",
+ "source": {
+ "name": "CML-TTS (LibriVox-derived, ylacombe/cml-tts on Hugging Face)",
+ "url": "https://huggingface.co/datasets/ylacombe/cml-tts",
+ "license": "CC-BY-4.0",
+ "consent": "LibriVox recording, redistributed under CC-BY; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "ines.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/ines.opus",
+ "sha256": "59f2fde8e98c8183e995e2576128ac4bbb186c08f677c1ac6daf2cd359bba7f4",
+ "duration_s": 10.44,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/ines.safetensors",
+ "sha256": "b7f76b52170dfcdcef0e8835956967bf2fbc4973883b6e99d6b2be7fd4f3a496"
+ },
+ "speaker_similarity": 0.96,
+ "sample": {
+ "seed": 7,
+ "text": "O meu amigo Jacinto nasceu num palácio, com cento e nove contos de renda em terras de semeadura, de vinhedo, de cortiça e de olival.",
+ "work": "A Cidade e as Serras (Eca de Queiros) — public domain",
+ "audio": "audio/ines.opus",
+ "sha256": "55c9cb8a3716ccfce280c25b577e344363fe3978b35577ce8c20ee6d4f3869ea"
+ },
+ "notes": "CML-TTS is LibriVox-derived; accent is Brazilian Portuguese, unlike tugao (European)."
+ },
+ {
+ "name": "nils",
+ "language": "Swedish",
+ "language_id": "sv",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": null,
+ "source": {
+ "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)",
+ "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "nils.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/nils.opus",
+ "sha256": "26680c82a976fe2d6f3202a97b50499274f1f0c36d329cbe6cdd6c3ebac1f0f0",
+ "duration_s": 11.0,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/nils.safetensors",
+ "sha256": "be75721c99547ab3ecdf1f8629101e9fa3ea1d569fc6e1220b8e2f259fb49f4e"
+ },
+ "speaker_similarity": 0.92,
+ "sample": {
+ "seed": 7,
+ "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.",
+ "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain",
+ "audio": "audio/nils.opus",
+ "sha256": "61b704118c82878a218d0db2b0fd3da20a70d0d442e2a36cc064ccc1a7f0679b"
+ }
+ },
+ {
+ "name": "selma",
+ "language": "Swedish",
+ "language_id": "sv",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": null,
+ "source": {
+ "name": "NST Swedish speech database (Sprakbanken, National Library of Norway)",
+ "url": "https://www.nb.no/sprakbanken/en/resource-catalogue/oai-nb-no-sbr-17/",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0 by Sprakbanken; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "selma.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/selma.opus",
+ "sha256": "775d4db5a95ac9fea95b1f9b8d6a5b1917a3aefd466e153962ad195d83ed51d0",
+ "duration_s": 8.53,
+ "construction": "single continuous clip"
+ },
+ "profile": {
+ "hf_path": "voices/selma.safetensors",
+ "sha256": "d3ba9ab18c676db23028c9561ad79a183b35adae377d20d46f1d8f253ce6eeec"
+ },
+ "speaker_similarity": 0.925,
+ "sample": {
+ "seed": 7,
+ "text": "Det var en gång en pojke, som var så där en fjorton år gammal, lång och ranglig och linhårig.",
+ "work": "Nils Holgerssons underbara resa (Selma Lagerlof) — public domain",
+ "audio": "audio/selma.opus",
+ "sha256": "70f186995d8b6aac965662cdff346ec5211bef98276c7ef94a69c096dccd0782"
+ }
+ },
+ {
+ "name": "soren",
+ "language": "Danish",
+ "language_id": "da",
+ "gender": "M",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "37",
+ "source": {
+ "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)",
+ "url": "https://huggingface.co/datasets/alexandrainst/nst-da",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "soren.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/soren.opus",
+ "sha256": "a0d7510fa2de66b8bf656c5f2ff93ddf462878ce31799860ca242bcad64487fb",
+ "duration_s": 7.38,
+ "construction": "single continuous clip, close mic"
+ },
+ "profile": {
+ "hf_path": "voices/soren.safetensors",
+ "sha256": "30d9fe9ee02928cd65ef7f4cb286c401c1809f07c13c19de79ffa375a027412e"
+ },
+ "speaker_similarity": 0.893,
+ "sample": {
+ "seed": 7,
+ "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.",
+ "work": "Den grimme aelling (H. C. Andersen) — public domain",
+ "audio": "audio/soren.opus",
+ "sha256": "f9841dd5162d30df3e67aee92128bd3708f4eff6c3d1c5c177091765cf04af2c"
+ }
+ },
+ {
+ "name": "freja",
+ "language": "Danish",
+ "language_id": "da",
+ "gender": "F",
+ "donor": "anonymous (invented name)",
+ "speaker_id": "35",
+ "source": {
+ "name": "NST Danish speech database (alexandrainst/nst-da on Hugging Face)",
+ "url": "https://huggingface.co/datasets/alexandrainst/nst-da",
+ "license": "CC0-1.0",
+ "consent": "corpus released CC0; anonymous speaker id, invented voice name"
+ },
+ "reference": {
+ "source_filename": "freja.wav",
+ "published_in_model_repo": false,
+ "public_preview": "audio/refs/freja.opus",
+ "sha256": "078baf737c9acc6f567dd87a8b3eca897becb0e83d9209279bf8314830d2678a",
+ "duration_s": 8.78,
+ "construction": "single continuous clip, close mic"
+ },
+ "profile": {
+ "hf_path": "voices/freja.safetensors",
+ "sha256": "0a3f913b9b47c529393f843850228bd8eee6a883b4c138978fd3afa7758561e8"
+ },
+ "speaker_similarity": 0.907,
+ "sample": {
+ "seed": 7,
+ "text": "Der var så dejligt ude på landet; det var sommer, kornet stod gult, og havren grøn.",
+ "work": "Den grimme aelling (H. C. Andersen) — public domain",
+ "audio": "audio/freja.opus",
+ "sha256": "cd785cc844250a5ae97010e2a3b0ee9260c5a24d016350d931aa7ec776958812"
+ }
+ }
+]