#!/usr/bin/env python3 """ OmniVoice Multilingual Voice Cloner Space: cosmichackerx/voice-cloner Zero-shot voice cloning & multilingual TTS powered by k2-fsa/OmniVoice. Crash-hardened: long-form, multi-speaker, voice profiles, BGM ducking, RVC-inspired polish (F0 / timbre / protect), OmniVoice advanced decoding. """ from __future__ import annotations import dataclasses import logging import os import re import tempfile import traceback from typing import Any import gradio as gr import numpy as np import spaces import torch import audio_enhance as ax logging.basicConfig( level=logging.INFO, format="%(asctime)s %(name)s %(levelname)s: %(message)s", ) logger = logging.getLogger("voice-cloner") CHECKPOINT = os.environ.get("OMNIVOICE_MODEL", "k2-fsa/OmniVoice") MAX_CHARS_PER_CHUNK = 280 MAX_TOTAL_CHARS = 8000 MAX_CHUNKS = 24 MAX_DIALOGUE_TURNS = 20 CROSSFADE_MS = 40 SILENCE_GAP_MS = 180 FEATURED = [ "Auto", "English", "Spanish", "French", "German", "Portuguese", "Italian", "Russian", "Chinese", "Japanese", "Korean", "Standard Arabic", "Hindi", "Turkish", "Indonesian", "Malay", "Vietnamese", "Thai", "Bengali", "Persian", "Urdu", "Dutch", "Polish", "Ukrainian", "Swahili", ] STYLE_PRESETS: dict[str, str] = { "Neutral": "", "Laughing intro": "[laughter] ", "Sigh then speak": "[sigh] ", "Surprised": "[surprise-oh] ", "Questioning tone": "[question-en] ", "Confirmation tone": "[confirmation-en] ", "Dissatisfied": "[dissatisfaction-hnn] ", } NONVERBAL_HELP = ( "Inline tags: [laughter] [sigh] [confirmation-en] [question-en] " "[question-ah] [question-oh] [surprise-ah] [surprise-oh] [dissatisfaction-hnn]" ) # One-click demo scripts — upload voice → click prompt → generate TRY_PROMPTS: dict[str, dict[str, str]] = { "YouTube intro": { "text": ( "Hey everyone, welcome back to the channel. Today I'm breaking down " "something that actually changed how I create content — stay until the end." ), "language": "English", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "Audiobook narrator": { "text": ( "The old lighthouse stood against the storm, its lamp cutting through fog " "like a promise. She climbed the spiral stairs, heart steady, knowing the " "letter in her coat would change everything." ), "language": "English", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "Podcast cold open": { "text": ( "You're listening to Signal Check. I'm your host, and in the next twenty minutes " "we'll unpack the story nobody expected — and what it means for all of us." ), "language": "English", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "Product ad read": { "text": ( "Meet Clarity Buds — all-day comfort, studio-clean sound, and a battery that " "keeps up with your life. Free shipping this week. Tap the link and hear the difference." ), "language": "English", "mode": "Voice clone", "style": "Confirmation tone", "instruct": "", }, "Meditation guide": { "text": ( "Take a slow breath in… and gently release. Soften your shoulders. " "There is nowhere else you need to be. Just this moment, just this breath." ), "language": "English", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "News bulletin": { "text": ( "Good evening. In tonight's top story, researchers released new findings on " "multilingual speech synthesis, marking a major step for accessible AI voice technology worldwide." ), "language": "English", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "Customer support": { "text": ( "Thanks for calling. I can help with that. Please confirm the email on your account, " "and I'll walk you through the next steps carefully." ), "language": "English", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "Gaming character": { "text": ( "[surprise-oh] The gate is open! Stay close — if we lose the map now, " "the whole quest falls apart. Move!" ), "language": "English", "mode": "Voice clone", "style": "Surprised", "instruct": "", }, "Laugh + punchline": { "text": ( "[laughter] Okay, that was not in the script. But seriously — if your AI voice " "can laugh with you, you're already winning." ), "language": "English", "mode": "Voice clone", "style": "Laughing intro", "instruct": "", }, "Spanish demo": { "text": ( "Hola, este es un demo de clonación de voz. Puedo hablar con naturalidad " "en español usando solo unos segundos de audio de referencia." ), "language": "Spanish", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "French demo": { "text": ( "Bonjour ! Voici une démonstration de clonage vocal. Avec quelques secondes " "d'audio, on peut générer une voix naturelle en français." ), "language": "French", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "Chinese demo": { "text": "大家好,这是零样本声音克隆演示。只需几秒参考音频,就能用中文自然地说出新的内容。", "language": "Chinese", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "Arabic demo": { "text": "مرحبًا، هذا عرض لنسخ الصوت بالذكاء الاصطناعي. يمكنني التحدث بالعربية بصوت قريب من المرجع.", "language": "Standard Arabic", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "Hindi demo": { "text": "नमस्ते! यह एक मुफ़्त एआई वॉइस क्लोन डेमो है। कुछ सेकंड के ऑडियो से नई हिंदी स्पीच बनाई जा सकती है।", "language": "Hindi", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "Japanese demo": { "text": "こんにちは。これはゼロショット音声クローンのデモです。短い参考音声から、自然な日本語を生成できます。", "language": "Japanese", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, "Multi-speaker dialogue": { "text": ( "[Speaker 1]: Did you try the new free AI voice cloner yet?\n" "[Speaker 2]: Yes — I uploaded ten seconds of audio and it already sounds like me.\n" "[Speaker 1]: Perfect for podcasts and YouTube voiceovers." ), "language": "English", "mode": "Multi-speaker dialogue", "style": "Neutral", "instruct": "", }, "Voice design · British": { "text": ( "This voice was designed from text attributes only — no reference clip required. " "Ideal when you need a consistent narrator fast." ), "language": "English", "mode": "Voice design (no reference)", "style": "Neutral", "instruct": "female, young adult, low pitch, british accent", }, "Voice design · Whisper": { "text": "Come closer. Some secrets are meant to be whispered, not shouted.", "language": "English", "mode": "Voice design (no reference)", "style": "Neutral", "instruct": "male, middle-aged, low pitch, whisper", }, "Long-form sample": { "text": ( "Welcome to this long-form narration test. First, we set the scene with a calm introduction. " "Next, we explain why clean reference audio matters for voice cloning quality. " "Finally, we finish with a clear call to action: like the Space, try another prompt, " "and share your best clone with creators who need multilingual text to speech." ), "language": "English", "mode": "Voice clone", "style": "Neutral", "instruct": "", }, } SPEAKER_RE = re.compile( r"^\s*\[?\s*(speaker\s*[12]|s\s*[12]|spk\s*[12])\s*\]?\s*[:\-–—]\s*(.+)$", re.IGNORECASE, ) SENTENCE_SPLIT_RE = re.compile(r"(?<=[.!?。!?…])\s+|(?<=[\n])\s*") logger.info("Loading OmniVoice from %s ...", CHECKPOINT) from omnivoice import OmniVoice, OmniVoiceGenerationConfig, VoiceClonePrompt try: from omnivoice.utils.lang_map import LANG_NAMES, lang_display_name _ALL = sorted({lang_display_name(code) for code in LANG_NAMES}) except Exception: # noqa: BLE001 logger.warning("Language map unavailable; using featured list only.") _ALL = [] LANGUAGE_CHOICES = FEATURED + [name for name in _ALL if name not in FEATURED] model = OmniVoice.from_pretrained( CHECKPOINT, device_map="auto", dtype=torch.float16, load_asr=True, ) SAMPLE_RATE = int(model.sampling_rate) logger.info("Model ready (sample rate=%s).", SAMPLE_RATE) try: _GEN_FIELDS = {f.name for f in dataclasses.fields(OmniVoiceGenerationConfig)} except Exception: # noqa: BLE001 _GEN_FIELDS = { "num_step", "guidance_scale", "denoise", "preprocess_prompt", "postprocess_output", } def _safe_str(value: Any, default: str = "") -> str: try: if value is None: return default return str(value) except Exception: # noqa: BLE001 return default def _resolve_language(language: str | None) -> str | None: if not language or language == "Auto": return None return language def _to_gradio_audio(audio_np: np.ndarray) -> tuple[int, np.ndarray]: wav = ax.as_float_mono(audio_np) return SAMPLE_RATE, (wav * 32767.0).astype(np.int16) def _as_float_mono(audio_np: np.ndarray) -> np.ndarray: return ax.as_float_mono(audio_np) def _crossfade_concat(parts: list[np.ndarray], gap_ms: int = SILENCE_GAP_MS) -> np.ndarray: clean = [_as_float_mono(p) for p in parts if p is not None and np.asarray(p).size > 0] if not clean: return np.zeros(1, dtype=np.float32) if len(clean) == 1: return clean[0] gap = max(0, int(SAMPLE_RATE * gap_ms / 1000.0)) fade = max(0, int(SAMPLE_RATE * CROSSFADE_MS / 1000.0)) out = clean[0] for nxt in clean[1:]: if gap > 0: out = np.concatenate([out, np.zeros(gap, dtype=np.float32), nxt]) continue if fade <= 0 or out.size < fade or nxt.size < fade: out = np.concatenate([out, nxt]) continue fade_out = np.linspace(1.0, 0.0, fade, dtype=np.float32) fade_in = np.linspace(0.0, 1.0, fade, dtype=np.float32) mixed = out[-fade:] * fade_out + nxt[:fade] * fade_in out = np.concatenate([out[:-fade], mixed, nxt[fade:]]) return out def _chunk_text(text: str, max_chars: int = MAX_CHARS_PER_CHUNK) -> list[str]: text = re.sub(r"\s+", " ", text.strip()) if not text: return [] if len(text) <= max_chars: return [text] pieces = [p.strip() for p in SENTENCE_SPLIT_RE.split(text) if p and p.strip()] or [text] chunks: list[str] = [] buf = "" for piece in pieces: if len(piece) > max_chars: if buf: chunks.append(buf) buf = "" for i in range(0, len(piece), max_chars): chunks.append(piece[i : i + max_chars].strip()) continue candidate = f"{buf} {piece}".strip() if buf else piece if len(candidate) <= max_chars: buf = candidate else: if buf: chunks.append(buf) buf = piece if buf: chunks.append(buf) return [c for c in chunks if c][:MAX_CHUNKS] def _parse_dialogue(text: str) -> list[tuple[int, str]] | None: lines = [ln.strip() for ln in text.splitlines() if ln.strip()] if len(lines) < 2: return None turns: list[tuple[int, str]] = [] tagged = 0 for ln in lines: m = SPEAKER_RE.match(ln) if m: tagged += 1 label = re.sub(r"\s+", "", m.group(1).lower()) speaker = 2 if label[-1] == "2" else 1 turns.append((speaker, m.group(2).strip())) else: turns.append((turns[-1][0] if turns else 1, ln)) if tagged == 0: return None return turns[:MAX_DIALOGUE_TURNS] def _apply_style(text: str, style: str | None) -> str: prefix = STYLE_PRESETS.get(_safe_str(style, "Neutral"), "") if not prefix or text.lstrip().startswith("["): return text return prefix + text def _load_audio_file(path: str | None) -> np.ndarray | None: if not path or not os.path.isfile(path): return None try: import soundfile as sf data, sr = sf.read(path, always_2d=False) wav = np.asarray(data, dtype=np.float32) if wav.ndim > 1: wav = wav.mean(axis=-1) if int(sr) != SAMPLE_RATE and wav.size > 0: try: import librosa wav = librosa.resample(wav, orig_sr=int(sr), target_sr=SAMPLE_RATE) except Exception: # noqa: BLE001 duration = wav.size / float(sr) new_len = max(1, int(duration * SAMPLE_RATE)) x_old = np.linspace(0.0, 1.0, num=wav.size, endpoint=False) x_new = np.linspace(0.0, 1.0, num=new_len, endpoint=False) wav = np.interp(x_new, x_old, wav).astype(np.float32) return _as_float_mono(wav) except Exception: # noqa: BLE001 logger.exception("Failed to load audio: %s", path) return None def _mix_with_bgm( voice: np.ndarray, bgm_path: str | None, bgm_volume: float = 0.18, duck_db: float = 8.0, ) -> np.ndarray: try: voice = _as_float_mono(voice) if not bgm_path: return voice bgm = _load_audio_file(bgm_path) if bgm is None or bgm.size == 0: return voice bgm_volume = float(np.clip(bgm_volume, 0.0, 1.0)) duck_gain = 10 ** (-float(np.clip(duck_db, 0.0, 24.0)) / 20.0) if bgm.size < voice.size: bgm = np.tile(bgm, int(np.ceil(voice.size / bgm.size)))[: voice.size] else: bgm = bgm[: voice.size] frame = max(1, int(SAMPLE_RATE * 0.02)) energy = np.convolve(np.abs(voice), np.ones(frame) / frame, mode="same") thresh = max(0.02, float(np.percentile(energy, 55)) * 0.6) gain = np.where(energy > thresh, duck_gain, 1.0).astype(np.float32) win = max(1, int(SAMPLE_RATE * 0.05)) gain = np.convolve(gain, np.ones(win) / win, mode="same") mixed = voice + bgm * bgm_volume * gain peak = float(np.max(np.abs(mixed))) if peak > 0.98: mixed = mixed * (0.98 / peak) return mixed.astype(np.float32) except Exception: # noqa: BLE001 logger.exception("BGM mix failed; returning dry voice") return _as_float_mono(voice) def _build_gen_config( num_step: float, guidance_scale: float, denoise: bool, preprocess_prompt: bool, postprocess_output: bool, t_shift: float | None = 0.1, position_temperature: float | None = 5.0, class_temperature: float | None = 0.0, pad_duration: float | None = 0.1, fade_duration: float | None = 0.1, audio_chunk_duration: float | None = 15.0, audio_chunk_threshold: float | None = 30.0, use_native_longform: bool = False, ) -> OmniVoiceGenerationConfig: steps = max(8, min(64, int(num_step or 32))) cfg = max(0.5, min(3.5, float(guidance_scale if guidance_scale is not None else 2.0))) payload: dict[str, Any] = { "num_step": steps, "guidance_scale": cfg, "denoise": bool(denoise), "preprocess_prompt": bool(preprocess_prompt), "postprocess_output": bool(postprocess_output), "t_shift": float(t_shift if t_shift is not None else 0.1), "position_temperature": float(position_temperature if position_temperature is not None else 5.0), "class_temperature": float(class_temperature if class_temperature is not None else 0.0), "pad_duration": float(pad_duration if pad_duration is not None else 0.1), "fade_duration": float(fade_duration if fade_duration is not None else 0.1), } if use_native_longform: payload["audio_chunk_duration"] = float(audio_chunk_duration or 15.0) payload["audio_chunk_threshold"] = float(audio_chunk_threshold or 30.0) filtered = {k: v for k, v in payload.items() if k in _GEN_FIELDS} try: return OmniVoiceGenerationConfig(**filtered) except Exception: # noqa: BLE001 logger.exception("GenConfig advanced fields failed; using minimal config") return OmniVoiceGenerationConfig( num_step=steps, guidance_scale=cfg, denoise=bool(denoise), preprocess_prompt=bool(preprocess_prompt), postprocess_output=bool(postprocess_output), ) def _make_prompt(ref_audio: str | None, ref_text: str | None) -> VoiceClonePrompt | None: if not ref_audio: return None try: return model.create_voice_clone_prompt( ref_audio=ref_audio, ref_text=(ref_text.strip() if ref_text and str(ref_text).strip() else None), ) except Exception: # noqa: BLE001 logger.exception("Failed to create voice clone prompt") return None def _generate_one( text: str, language: str | None, gen_config: OmniVoiceGenerationConfig, prompt: VoiceClonePrompt | None, speed: float, duration: float | None, instruct: str | None, normalize_text: bool, blend_instruct: bool = False, ) -> np.ndarray | None: text = text.strip() if not text: return None kw: dict[str, Any] = { "text": text, "language": language, "generation_config": gen_config, } if prompt is not None: kw["voice_clone_prompt"] = prompt # OmniVoice tip: consistent instruct + ref improves dialect/style stability if blend_instruct and instruct and str(instruct).strip(): kw["instruct"] = str(instruct).strip() elif instruct and str(instruct).strip(): kw["instruct"] = str(instruct).strip() if speed is not None and abs(float(speed) - 1.0) > 1e-3: kw["speed"] = float(np.clip(speed, 0.7, 1.3)) if duration is not None and float(duration) > 0: kw["duration"] = float(duration) if normalize_text: kw["normalize_text"] = True try: audio = model.generate(**kw) if not audio: return None return _as_float_mono(audio[0]) except TypeError: for drop in ("normalize_text", "instruct"): kw.pop(drop, None) try: audio = model.generate(**kw) if audio: return _as_float_mono(audio[0]) except Exception: # noqa: BLE001 continue logger.exception("Generation failed for chunk") return None except Exception: # noqa: BLE001 logger.exception("Generation failed for chunk") return None def _export_audio(wav: np.ndarray, format_type: str = "WAV") -> str | None: try: import soundfile as sf import tempfile ext = f".{format_type.lower()}" tmp = tempfile.NamedTemporaryFile(delete=False, suffix=ext, prefix="clone_") tmp.close() fmt = format_type.upper() if fmt == "MP3": try: sf.write(tmp.name, wav, SAMPLE_RATE, format="MP3") except Exception: import scipy.io.wavfile as wavfile from pydub import AudioSegment import os tmp_wav = tmp.name + ".wav" wavfile.write(tmp_wav, SAMPLE_RATE, wav) audio = AudioSegment.from_wav(tmp_wav) audio.export(tmp.name, format="mp3") os.remove(tmp_wav) else: sf.write(tmp.name, wav, SAMPLE_RATE, format=fmt) return tmp.name except Exception as e: print(f"Export error: {e}") return None def update_export(audio_data, format_type): if audio_data is None: return None sr, y = audio_data return _export_audio(y, format_type) def _finalize( parts: list[np.ndarray], bgm_audio, bgm_volume, duck_db, ref_for_polish: np.ndarray | None, f0_semitones, index_rate, protect, loud_norm, enable_polish: bool, download_format: str = "WAV", ) -> tuple[np.ndarray, str, str | None]: voice = _crossfade_concat(parts) polish_note = "Polish: off" if enable_polish: voice, polish_note = ax.apply_rvc_inspired_polish( voice, ref_for_polish, SAMPLE_RATE, f0_semitones=float(f0_semitones or 0), index_rate=float(index_rate or 0), protect=float(protect or 0), loud_norm=bool(loud_norm), ) voice = _mix_with_bgm(voice, bgm_audio, float(bgm_volume or 0.18), float(duck_db or 8)) return voice, polish_note, _export_audio(voice, download_format) def _estimate_gpu_seconds( text, language, ref_audio, ref_text, num_step, guidance_scale, denoise, speed, duration, preprocess_prompt, postprocess_output, style, long_form, mode, ref_audio_2, ref_text_2, instruct, normalize_text, bgm_audio, bgm_volume, duck_db, *args, **kwargs, ) -> int: try: raw = _safe_str(text) n_chars = min(len(raw), MAX_TOTAL_CHARS) dialogue = _parse_dialogue(raw) if mode == "Multi-speaker dialogue" else None if dialogue: n_chunks = min(len(dialogue), MAX_DIALOGUE_TURNS) elif long_form or n_chars > 1200: n_chunks = max(1, min(MAX_CHUNKS, (n_chars // MAX_CHARS_PER_CHUNK) + 1)) else: n_chunks = 1 steps = int(num_step or 32) per = 8 + int(steps * 0.35) # Pitch/timbre polish is CPU — small buffer only total = 30 + n_chunks * per return int(min(280, max(45, total))) except Exception: # noqa: BLE001 return 120 # --------------------------------------------------------------------------- # Voice library # --------------------------------------------------------------------------- def save_voice_profile(name, ref_audio, ref_text, library): library = dict(library or {}) name = _safe_str(name).strip() if not name: return library, gr.update(), "Enter a profile name." if not ref_audio: return library, gr.update(), "Upload reference audio before saving a profile." if len(library) >= 12: return library, gr.update(choices=list(library.keys())), "Library full (max 12)." prompt = _make_prompt(ref_audio, ref_text) if prompt is None: return library, gr.update(choices=list(library.keys())), "Could not encode reference." try: tmp = tempfile.NamedTemporaryFile(delete=False, suffix=".pt", prefix=f"voice_{name}_") tmp.close() prompt.save(tmp.name) library[name] = {"path": tmp.name, "ref_text": _safe_str(ref_text).strip(), "ref_audio": ref_audio} return library, gr.update(choices=list(library.keys()), value=name), f"Saved “{name}”." except Exception as exc: # noqa: BLE001 return library, gr.update(choices=list(library.keys())), f"Save failed: {exc}" def load_voice_profile(name, library): library = library or {} name = _safe_str(name).strip() if not name or name not in library: return None, "", None, "Select a saved profile." entry = library[name] path = entry.get("path") return entry.get("ref_audio"), entry.get("ref_text", ""), (path if path and os.path.isfile(path) else None), f"Loaded “{name}”." def import_voice_pt(pt_file, name, library): library = dict(library or {}) name = _safe_str(name).strip() or "Imported" if not pt_file or not os.path.isfile(pt_file): return library, gr.update(), "Upload a .pt voice prompt file." try: _ = VoiceClonePrompt.load(pt_file) tmp = tempfile.NamedTemporaryFile(delete=False, suffix=".pt", prefix=f"voice_{name}_") tmp.close() with open(pt_file, "rb") as src, open(tmp.name, "wb") as dst: dst.write(src.read()) library[name] = {"path": tmp.name, "ref_text": "", "ref_audio": None} return library, gr.update(choices=list(library.keys()), value=name), f"Imported “{name}”." except Exception as exc: # noqa: BLE001 return library, gr.update(choices=list(library.keys())), f"Import failed: {exc}" def delete_voice_profile(name, library): library = dict(library or {}) name = _safe_str(name).strip() if name in library: try: path = library[name].get("path") if path and os.path.isfile(path): os.remove(path) except Exception: # noqa: BLE001 pass del library[name] return library, gr.update(choices=list(library.keys()), value=None), f"Deleted “{name}”." return library, gr.update(choices=list(library.keys())), "Nothing to delete." def _prompt_from_library_or_audio(profile_name, library, ref_audio, ref_text): library = library or {} name = _safe_str(profile_name).strip() if name and name in library: path = library[name].get("path") if path and os.path.isfile(path): try: return VoiceClonePrompt.load(path), f"Using saved profile “{name}”.", library[name].get("ref_audio") except Exception: # noqa: BLE001 logger.exception("Failed loading profile %s", name) if ref_audio: prompt = _make_prompt(ref_audio, ref_text) if prompt is not None: return prompt, "Using uploaded reference audio.", ref_audio return None, "Could not encode reference audio.", None return None, "No reference voice provided.", None def apply_quality_preset(preset: str): cfg = ax.QUALITY_PRESETS.get(preset, ax.QUALITY_PRESETS["Balanced"]) return cfg["num_step"], cfg["guidance_scale"], cfg["t_shift"] def analyze_ref_ui(ref_audio): return ax.analyze_reference(ref_audio, SAMPLE_RATE) def apply_try_prompt(prompt_name: str): """Fill text / language / mode / style / instruct from a curated try prompt.""" entry = TRY_PROMPTS.get(_safe_str(prompt_name)) if not entry: return gr.update(), gr.update(), gr.update(), gr.update(), gr.update(), "Pick a try prompt." return ( entry["text"], entry.get("language", "Auto"), entry.get("mode", "Voice clone"), entry.get("style", "Neutral"), entry.get("instruct", ""), f"Loaded “{prompt_name}”. Upload/record your voice → enable consent → Generate.", ) # --------------------------------------------------------------------------- # Core generation # --------------------------------------------------------------------------- @spaces.GPU(duration=_estimate_gpu_seconds) def clone_voice( text, language, ref_audio, ref_text, num_step, guidance_scale, denoise, speed, duration, preprocess_prompt, postprocess_output, style, long_form, mode, ref_audio_2, ref_text_2, instruct, normalize_text, bgm_audio, bgm_volume, duck_db, profile_name, library, consent, ref_prep, blend_instruct, use_native_longform, t_shift, position_temperature, class_temperature, pad_duration, fade_duration, audio_chunk_duration, audio_chunk_threshold, enable_polish, f0_semitones, index_rate, protect, loud_norm, download_format, progress=gr.Progress(track_tqdm=False), ): try: yield from _clone_voice_impl( text, language, ref_audio, ref_text, num_step, guidance_scale, denoise, speed, duration, preprocess_prompt, postprocess_output, style, long_form, mode, ref_audio_2, ref_text_2, instruct, normalize_text, bgm_audio, bgm_volume, duck_db, profile_name, library, consent, ref_prep, blend_instruct, use_native_longform, t_shift, position_temperature, class_temperature, pad_duration, fade_duration, audio_chunk_duration, audio_chunk_threshold, enable_polish, f0_semitones, index_rate, protect, loud_norm, download_format, progress, ) except Exception as exc: # noqa: BLE001 logger.error("Unhandled clone_voice error:\n%s", traceback.format_exc()) yield None, f"Unexpected error (recovered): {type(exc).__name__}: {exc}", None def _clone_voice_impl( text, language, ref_audio, ref_text, num_step, guidance_scale, denoise, speed, duration, preprocess_prompt, postprocess_output, style, long_form, mode, ref_audio_2, ref_text_2, instruct, normalize_text, bgm_audio, bgm_volume, duck_db, profile_name, library, consent, ref_prep, blend_instruct, use_native_longform, t_shift, position_temperature, class_temperature, pad_duration, fade_duration, audio_chunk_duration, audio_chunk_threshold, enable_polish, f0_semitones, index_rate, protect, loud_norm, download_format, progress, ): if not consent: yield None, "Please confirm you have rights/consent to clone this voice.", None return raw = _safe_str(text).strip() if not raw: yield None, "Enter text to speak in any of 600+ supported languages.", None return if len(raw) > MAX_TOTAL_CHARS: yield None, f"Text too long. Keep under {MAX_TOTAL_CHARS} characters.", None return mode = _safe_str(mode, "Voice clone") lang = _resolve_language(language) gen_config = _build_gen_config( num_step, guidance_scale, denoise, preprocess_prompt, postprocess_output, t_shift, position_temperature, class_temperature, pad_duration, fade_duration, audio_chunk_duration, audio_chunk_threshold, bool(use_native_longform), ) speed_f = float(speed) if speed is not None else 1.0 dur_f = float(duration) if duration is not None and float(duration or 0) > 0 else None use_duration = dur_f if (not long_form and mode != "Multi-speaker dialogue") else None blend = bool(blend_instruct) polish_on = bool(enable_polish) # Optional reference preparation (RVC/OmniVoice quality tip: clean short ref) prep_notes = [] if bool(ref_prep) and ref_audio: ref_audio, note = ax.prepare_reference(ref_audio, SAMPLE_RATE, enable=True, gate=True) prep_notes.append(note) if bool(ref_prep) and ref_audio_2: ref_audio_2, note2 = ax.prepare_reference(ref_audio_2, SAMPLE_RATE, enable=True, gate=True) prep_notes.append("S2: " + note2) def _done(parts, ref_path, msg): ref_wav = _load_audio_file(ref_path) if ref_path else None final, polish_note, wav_path = _finalize( parts, bgm_audio, bgm_volume, duck_db, ref_wav, f0_semitones, index_rate, protect, loud_norm, polish_on, download_format ) extra = " · ".join([x for x in prep_notes + [polish_note] if x]) if bgm_audio: msg += " · BGM ducked" if extra: msg += f" · {extra}" yield _to_gradio_audio(final), msg, wav_path # -------- Voice design -------- if mode == "Voice design (no reference)": instr = _safe_str(instruct).strip() if not instr: yield None, "Voice design needs instruct, e.g. “female, low pitch, british accent”.", None return styled = _apply_style(raw, style) # Prefer native OmniVoice chunking when enabled; else our splitter if use_native_longform and not long_form: chunks = [styled] else: chunks = _chunk_text(styled) if (long_form or len(styled) > 1200) else [styled] yield None, f"Voice design · {len(chunks)} chunk(s)…", None parts: list[np.ndarray] = [] for i, chunk in enumerate(chunks): progress((i + 1) / len(chunks), desc=f"Design {i + 1}/{len(chunks)}") wav = _generate_one( chunk, lang, gen_config, None, speed_f, use_duration if len(chunks) == 1 else None, instr, bool(normalize_text), False, ) if wav is None: if parts: yield from _done(parts, None, f"Stopped at chunk {i + 1} (partial kept).") else: yield None, f"Failed at chunk {i + 1}.", None return parts.append(wav) yield _to_gradio_audio(_crossfade_concat(parts)), f"Chunk {i + 1}/{len(chunks)}…", None yield from _done(parts, None, "Done — voice design ready.") return if mode == "Auto voice (random)": styled = _apply_style(raw, style) chunks = [styled] if (use_native_longform and not long_form) else ( _chunk_text(styled) if (long_form or len(styled) > 1200) else [styled] ) yield None, f"Auto voice · {len(chunks)} chunk(s)…", None parts = [] for i, chunk in enumerate(chunks): progress((i + 1) / len(chunks), desc=f"Auto {i + 1}/{len(chunks)}") wav = _generate_one( chunk, lang, gen_config, None, speed_f, use_duration if len(chunks) == 1 else None, None, bool(normalize_text), False, ) if wav is None: if parts: yield from _done(parts, None, f"Stopped at chunk {i + 1} (partial kept).") else: yield None, f"Failed at chunk {i + 1}.", None return parts.append(wav) yield _to_gradio_audio(_crossfade_concat(parts)), f"Chunk {i + 1}/{len(chunks)}…", None yield from _done(parts, None, "Done — auto voice ready.") return # -------- Multi-speaker -------- if mode == "Multi-speaker dialogue": turns = _parse_dialogue(raw) if not turns: yield None, "Use lines like:\n[Speaker 1]: Hello!\n[Speaker 2]: Hi!", None return if not ref_audio and not (profile_name and library and _safe_str(profile_name) in (library or {})): yield None, "Speaker 1 needs a reference clip (or saved profile).", None return if not ref_audio_2: yield None, "Speaker 2 needs a second reference clip.", None return p1, note1, ref1_path = _prompt_from_library_or_audio(profile_name, library, ref_audio, ref_text) p2 = _make_prompt(ref_audio_2, ref_text_2) if p1 is None or p2 is None: yield None, f"Could not encode speaker prompts. {note1}", None return yield None, f"Dialogue · {len(turns)} turns. {note1}", None parts = [] for i, (spk, line) in enumerate(turns): progress((i + 1) / len(turns), desc=f"Turn {i + 1}/{len(turns)}") prompt = p1 if spk == 1 else p2 wav = _generate_one( _apply_style(line, style), lang, gen_config, prompt, speed_f, None, instruct if blend else None, bool(normalize_text), blend, ) if wav is None: if parts: yield from _done(parts, ref1_path, f"Stopped at turn {i + 1} (partial kept).") else: yield None, f"Failed at turn {i + 1}.", None return parts.append(wav) yield _to_gradio_audio(_crossfade_concat(parts)), f"Turn {i + 1}/{len(turns)}…", None yield from _done(parts, ref1_path, "Done — multi-speaker dialogue ready.") return # -------- Standard / long-form clone -------- prompt, note, ref_path = _prompt_from_library_or_audio(profile_name, library, ref_audio, ref_text) if prompt is None: yield None, note or "Upload a clear 3–10 second reference voice clip.", None return styled = _apply_style(raw, style) do_long = bool(long_form) or (len(styled) > 1200 and not use_native_longform) if len(styled) > 1200 and not do_long and not use_native_longform: yield None, "Text over ~1200 chars — enable Long-form or Native OmniVoice chunking.", None return if use_native_longform and not do_long: chunks = [styled] # let OmniVoice split internally else: chunks = _chunk_text(styled) if do_long else [styled] if not chunks: yield None, "Nothing to speak after preprocessing.", None return yield None, f"{note} Generating {len(chunks)} chunk(s)…", None parts = [] for i, chunk in enumerate(chunks): progress((i + 1) / len(chunks), desc=f"Chunk {i + 1}/{len(chunks)}") wav = _generate_one( chunk, lang, gen_config, prompt, speed_f, use_duration if len(chunks) == 1 else None, instruct if blend else None, bool(normalize_text), blend, ) if wav is None: if parts: yield from _done(parts, ref_path, f"Stopped at chunk {i + 1} (partial kept).") else: yield None, f"Failed at chunk {i + 1}.", None return parts.append(wav) yield _to_gradio_audio(_crossfade_concat(parts)), f"Chunk {i + 1}/{len(chunks)}…", None msg = "Done — cloned speech ready." if len(chunks) > 1: msg = f"Done — long-form clone ({len(chunks)} chunks)." yield from _done(parts, ref_path, msg) # --------------------------------------------------------------------------- # UI # --------------------------------------------------------------------------- CUSTOM_CSS = """ /* Base Container */ .gradio-container { max-width: 95% !important; /* Expanded for landscape */ font-family: 'Inter', sans-serif !important; } /* Glassmorphism Panels */ .glass-panel { background: rgba(30, 41, 59, 0.95) !important; border: 1px solid rgba(255, 255, 255, 0.1) !important; border-radius: 16px !important; box-shadow: 0 8px 32px 0 rgba(0, 0, 0, 0.3) !important; padding: 32px !important; margin-bottom: 24px !important; } /* Checkbox & Input Polish */ label.checkbox { font-size: 1.2em !important; font-weight: 700 !important; color: #ffffff !important; letter-spacing: 0.5px !important; } label.checkbox input { transform: scale(1.5) !important; margin-right: 14px !important; cursor: pointer !important; } .gr-box, .gr-accordion, .gr-input, .gr-dropdown, .gr-radio { font-size: 1.05em !important; } #vertical-mode .wrap, #vertical-mode fieldset { flex-direction: column !important; align-items: flex-start !important; } #vertical-mode label { margin-bottom: 8px !important; } /* Typography */ h1 { font-weight: 800 !important; background: -webkit-linear-gradient(45deg, #4ade80, #3b82f6); -webkit-background-clip: text; -webkit-text-fill-color: transparent; text-align: center; margin-bottom: 10px !important; } .subtitle { text-align: center; color: #94a3b8 !important; font-size: 1.1em; margin-bottom: 30px !important; } /* Generate Button */ #clone-btn { background: linear-gradient(90deg, #10b981 0%, #3b82f6 100%) !important; border: none !important; color: white !important; font-weight: 700 !important; font-size: 1.2rem !important; border-radius: 12px !important; padding: 16px !important; transition: all 0.3s ease !important; box-shadow: 0 4px 15px rgba(16, 185, 129, 0.4) !important; } #clone-btn:hover { transform: translateY(-2px) !important; box-shadow: 0 6px 20px rgba(16, 185, 129, 0.6) !important; } /* Analyze Button */ #analyze-btn { background: linear-gradient(90deg, #3b82f6 0%, #6366f1 100%) !important; border: none !important; color: white !important; font-weight: 600 !important; border-radius: 8px !important; transition: all 0.3s ease !important; width: 100% !important; max-height: 42px !important; padding: 8px 16px !important; white-space: nowrap !important; display: flex !important; justify-content: center !important; align-items: center !important; line-height: 1 !important; } .align-center-row { align-items: center !important; } #analyze-btn:hover { transform: translateY(-1px) !important; box-shadow: 0 4px 15px rgba(59, 130, 246, 0.5) !important; } .gr-box { border-radius: 12px !important; } .gr-accordion { border: 1px solid rgba(255, 255, 255, 0.05) !important; border-radius: 12px !important; background: rgba(15, 23, 42, 0.6) !important; } """ with gr.Blocks( title="AI Voice Cloner — Free Zero-Shot TTS (600+ Languages) | OmniVoice", theme=gr.themes.Base( primary_hue="emerald", secondary_hue="blue", neutral_hue="slate", font=[gr.themes.GoogleFont("Inter"), "system-ui", "sans-serif"], ), css=CUSTOM_CSS, ) as demo: voice_library = gr.State({}) gr.HTML( """
Free online AI voice cloning & multilingual text-to-speech. Upload 3–10 seconds of audio → generate natural speech in 600+ languages.
Zero-shot voice clone · cross-lingual TTS · long-form · multi-speaker · RVC-inspired polish · one-click try prompts.
Powered by k2-fsa/OmniVoice on ZeroGPU · Weights CC-BY-NC · ⭐ Like this Space to boost discovery