| """Bedtime narration: Kokoro-82M (Apache 2.0) — an 82M-parameter TTS that fits the |
| hackathon's small-model spirit perfectly. |
| |
| Voices: zh / en / fr supported; German has no Kokoro voice -> returns None and the |
| UI degrades gracefully to text-only for German stories. |
| """ |
|
|
| from __future__ import annotations |
|
|
| import os |
| from typing import Optional |
|
|
| import numpy as np |
|
|
| MOCK = os.environ.get("STORY_MOCK") == "1" |
|
|
| SAMPLE_RATE = 24_000 |
|
|
| |
| VOICES = { |
| "zh": ("z", "zf_xiaobei"), |
| "en": ("a", "af_heart"), |
| "fr": ("f", "ff_siwis"), |
| } |
|
|
| _pipelines: dict[str, object] = {} |
|
|
|
|
| def _get_pipeline(lang_code: str): |
| if lang_code not in _pipelines: |
| from kokoro import KPipeline |
|
|
| |
| |
| |
| _pipelines[lang_code] = KPipeline( |
| lang_code=lang_code, repo_id="hexgrad/Kokoro-82M", device="cpu" |
| ) |
| return _pipelines[lang_code] |
|
|
|
|
| def synthesize(text: str, language: str) -> Optional[tuple[int, np.ndarray]]: |
| """Returns (sample_rate, waveform) for gr.Audio, or None when unsupported (de).""" |
| if language not in VOICES: |
| return None |
| if MOCK: |
| return SAMPLE_RATE, np.zeros(SAMPLE_RATE // 2, dtype=np.float32) |
|
|
| lang_code, voice = VOICES[language] |
| pipeline = _get_pipeline(lang_code) |
| chunks: list[np.ndarray] = [] |
| pause = np.zeros(int(SAMPLE_RATE * 0.45), dtype=np.float32) |
| for _, _, audio in pipeline(text, voice=voice): |
| wav = audio.numpy() if hasattr(audio, "numpy") else np.asarray(audio) |
| chunks.append(wav.astype(np.float32)) |
| chunks.append(pause) |
| if not chunks: |
| return None |
| return SAMPLE_RATE, np.concatenate(chunks) |
|
|