| import gradio as gr |
| import torch |
| import soundfile as sf |
| import tempfile |
| import os |
| from omnivoice import OmniVoice |
|
|
| |
| device = "cuda" if torch.cuda.is_available() else "cpu" |
| dtype = torch.float16 if device == "cuda" else torch.float32 |
|
|
| model = OmniVoice.from_pretrained( |
| "kjanh/KhanhTTS-OmniVoice", |
| device_map=device, |
| dtype=dtype |
| ) |
|
|
| def synthesize(text, ref_audio=None, ref_text=None): |
| if not text.strip(): |
| return None, "❌ Vui lòng nhập văn bản." |
| |
| kwargs = {"text": text} |
| if ref_audio is not None: |
| kwargs["ref_audio"] = ref_audio |
| if ref_text and ref_text.strip(): |
| kwargs["ref_text"] = ref_text.strip() |
| |
| audio_list = model.generate(**kwargs) |
| audio = audio_list[0] |
| |
| tmp = tempfile.NamedTemporaryFile(suffix=".wav", delete=False) |
| sf.write(tmp.name, audio, 24000) |
| return tmp.name, "✅ Thành công!" |
|
|
| with gr.Blocks(title="KhanhTTS - Voice Clone TTS") as demo: |
| gr.Markdown("# 🗣️ KhanhTTS — Tiếng Việt & Tiếng Anh") |
| gr.Markdown("Model: `kjanh/KhanhTTS-OmniVoice` · Fine-tuned trên ~1500h audio Việt+Anh") |
| |
| with gr.Row(): |
| with gr.Column(): |
| text_input = gr.Textbox( |
| label="Văn bản cần đọc", |
| placeholder="Nhập văn bản tiếng Việt hoặc tiếng Anh...", |
| lines=4 |
| ) |
| ref_audio_input = gr.Audio( |
| label="🎙️ Giọng tham chiếu (tuỳ chọn — để clone giọng)", |
| type="filepath" |
| ) |
| ref_text_input = gr.Textbox( |
| label="Transcript của giọng tham chiếu (tuỳ chọn)", |
| placeholder="Nội dung lời nói trong file audio tham chiếu...", |
| lines=2 |
| ) |
| btn = gr.Button("🔊 Tổng hợp giọng nói", variant="primary") |
| |
| with gr.Column(): |
| audio_output = gr.Audio(label="🔈 Kết quả") |
| status = gr.Textbox(label="Trạng thái", interactive=False) |
| |
| btn.click( |
| fn=synthesize, |
| inputs=[text_input, ref_audio_input, ref_text_input], |
| outputs=[audio_output, status] |
| ) |
| |
| gr.Examples( |
| examples=[ |
| ["Xin chào các bạn, đây là demo tổng hợp giọng nói tiếng Việt.", None, None], |
| ["Hello, this is a demonstration of Vietnamese and English TTS.", None, None], |
| ], |
| inputs=[text_input, ref_audio_input, ref_text_input] |
| ) |
|
|
| demo.launch() |