File size: 5,932 Bytes
0ab8461
 
 
eef1644
0ab8461
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
eef1644
0ab8461
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
ba58530
783eed5
 
 
 
ba58530
 
 
 
 
783eed5
 
 
ba58530
 
 
0ab8461
783eed5
ba58530
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0ab8461
 
ba58530
 
 
 
 
 
 
0ab8461
 
 
 
 
ba58530
0ab8461
 
 
ba58530
 
 
 
 
 
0ab8461
 
ba58530
bb72a46
ba58530
 
 
 
0ab8461
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
import os
import gradio as gr
import soundfile as sf
import spaces
from huggingface_hub import snapshot_download

MODEL_REPO = "LyngualLabs/YorubaEnglish-CodeSwitching-TTS"
DEFAULT_REF_WAV = "reference.wav"
DEFAULT_REF_TEXT = "Ìròyìn ti sọ pé the government will ensure electricity tariff goes down ní January."

print("Downloading model...", flush=True)
model_dir = snapshot_download(MODEL_REPO)

from voxcpm import VoxCPM  # noqa: E402  (import after snapshot_download so the repo is cached)

print("Loading model...", flush=True)
model = VoxCPM.from_pretrained(model_dir, load_denoiser=False)
SR = model.tts_model.sample_rate
print("Model loaded.", flush=True)


@spaces.GPU(duration=120)
def synthesize(text, ref_audio, ref_text, cfg_value, inference_timesteps):
    if not text or not text.strip():
        raise gr.Error("Please enter some text to synthesize.")

    prompt_wav_path = ref_audio if ref_audio else DEFAULT_REF_WAV
    prompt_text = ref_text.strip() if (ref_audio and ref_text and ref_text.strip()) else (
        DEFAULT_REF_TEXT if not ref_audio else ""
    )
    if ref_audio and not prompt_text:
        raise gr.Error("Please provide the exact transcript of your uploaded reference audio.")

    wav = model.generate(
        text=text.strip(),
        prompt_wav_path=prompt_wav_path,
        prompt_text=prompt_text,
        cfg_value=cfg_value,
        inference_timesteps=int(inference_timesteps),
    )
    out_path = "output.wav"
    sf.write(out_path, wav, SR)
    return out_path


EXAMPLES = [
    ["Mo ní meeting pẹ̀lú marketing team ní aago mẹ́wàá lónìí, nítorí náà ẹ jẹ́ ká yára."],
    ["Ìròyìn ti sọ pé the government will ensure electricity tariff goes down ní January."],
    ["Wọ́n ní ìpàdé ní hotel tó wà ní Victoria Island, kì í ṣe ní office wa."],
]

CSS = """
@import url('https://fonts.googleapis.com/css2?family=Inter:wght@400;600;700&display=swap');
:root, .dark { --tts-primary: #059669; --tts-primary-hover: #047857; --tts-accent: #b45309; }
.gradio-container { max-width: 960px !important; margin: auto !important;
                     font-family: 'Inter', system-ui, sans-serif !important; }
#header { text-align: center; padding: 0.5rem 0 0.25rem 0; }
#header h1 { font-size: 2.1rem; margin-bottom: 0.2rem; }
#subtitle { text-align: center; color: var(--body-text-color-subdued); font-size: 1.05rem;
            max-width: 640px; margin: 0 auto 0.75rem auto; }
#badges { display: flex; gap: 0.5rem; justify-content: center; flex-wrap: wrap; margin-bottom: 0.5rem; }
#generate-btn { font-size: 1.05rem !important; font-weight: 600 !important;
                background: var(--tts-primary) !important; border-color: var(--tts-primary) !important; }
#generate-btn:hover { background: var(--tts-primary-hover) !important; }
#footer { text-align: center; color: var(--body-text-color-subdued); font-size: 0.85rem;
          margin-top: 1rem; padding-top: 0.75rem; border-top: 1px solid var(--border-color-primary); }
"""

with gr.Blocks(title="Yoruba-English Code-Switching TTS", css=CSS) as demo:
    gr.Markdown("# 🎙️ Yoruba–English Code-Switching TTS", elem_id="header")
    gr.Markdown(
        "A voice-cloning text-to-speech model for Yoruba, including natural "
        "Yoruba–English code-switching — full fine-tuned from "
        "[VoxCPM2](https://huggingface.co/openbmb/VoxCPM2) on ~1,039 hours pooled from "
        "DSN African Voices, NaijaVoices, YECS, and WAXAL.",
        elem_id="subtitle",
    )
    gr.HTML(
        """
        <div id="badges">
            <a href="https://huggingface.co/LyngualLabs/YorubaEnglish-CodeSwitching-TTS" target="_blank">
                <img src="https://img.shields.io/badge/🤗%20Model-LyngualLabs-2e7d32?style=flat-square" />
            </a>
            <a href="https://huggingface.co/datasets/Professor/yoruba-speech-data" target="_blank">
                <img src="https://img.shields.io/badge/📊%20Dataset-1039h-b45309?style=flat-square" />
            </a>
        </div>
        """
    )
    gr.Markdown(
        "> ⚠️ **Experimental** — trained for only ~1 epoch. Quality varies by sentence; "
        "rarer words/proper nouns are more likely to be mispronounced."
    )

    with gr.Row(equal_height=False):
        with gr.Column(scale=1):
            text_in = gr.Textbox(
                label="Text to speak",
                placeholder="Type Yoruba, English, or mixed Yoruba-English text...",
                lines=4,
            )
            with gr.Accordion("🎭 Voice cloning — upload your own reference", open=False):
                ref_audio = gr.Audio(label="Reference audio (3-10s, clean)", type="filepath")
                ref_text = gr.Textbox(label="Exact transcript of the reference audio")
                gr.Markdown("_Leave blank to use the default built-in voice._")
            with gr.Accordion("⚙️ Advanced settings", open=False):
                cfg_value = gr.Slider(1.0, 4.0, value=2.0, step=0.1, label="CFG value (higher = sticks closer to reference)")
                inference_timesteps = gr.Slider(10, 35, value=22, step=1, label="Inference timesteps (higher = smoother, slower)")
            btn = gr.Button("✨ Generate Speech", variant="primary", elem_id="generate-btn", size="lg")
            gr.Examples(examples=EXAMPLES, inputs=[text_in], label="Try an example")
        with gr.Column(scale=1):
            audio_out = gr.Audio(label="Generated speech", type="filepath")

    gr.Markdown(
        "Part of the YECS-Downstream speech research effort by LyngualLabs · "
        "[Model card](https://huggingface.co/LyngualLabs/YorubaEnglish-CodeSwitching-TTS)",
        elem_id="footer",
    )

    btn.click(
        synthesize,
        inputs=[text_in, ref_audio, ref_text, cfg_value, inference_timesteps],
        outputs=[audio_out],
    )

if __name__ == "__main__":
    demo.launch()