voice-clone / app.py
2013khansohail's picture
Add voice cloning application
e25430f
Raw
History Blame Contribute Delete
4.39 kB
import gradio as gr
import torch
import os
import tempfile
import time
from TTS.api import TTS
# Global variable to store the TTS model
model = None
def load_model():
global model
try:
# Load model on first run
if model is None:
print("Loading XTTS model... (this may take a few minutes)")
model = TTS("tts_models/multilingual/multi-dataset/xtts_v2")
print("Model loaded successfully!")
return True
except Exception as e:
print(f"Error loading model: {str(e)}")
return False
def clone_voice(text, audio_file, progress=gr.Progress()):
"""
Generate speech using XTTS with the uploaded reference audio
"""
if not text or not audio_file:
return None, "Please provide both text and a voice sample."
progress(0.1, desc="Loading model...")
# Try to load the model (if not already loaded)
model_loaded = load_model()
if not model_loaded:
return None, "Failed to load the TTS model. This might be due to memory constraints."
progress(0.3, desc="Processing voice sample...")
try:
# Generate speech with the cloned voice
progress(0.6, desc="Generating speech (this may take a while on CPU)...")
# Create a temporary directory for output
temp_dir = tempfile.gettempdir()
output_path = os.path.join(temp_dir, "xtts_output.wav")
# Use the model to generate speech
start_time = time.time()
model.tts_to_file(
text=text,
file_path=output_path,
speaker_wav=audio_file,
language="en"
)
generation_time = time.time() - start_time
progress(1.0, desc="Done!")
return output_path, f"Generation completed in {generation_time:.1f} seconds."
except Exception as e:
error_message = str(e)
if "CUDA" in error_message:
return None, "CUDA error: This Space is running on CPU. Please keep your text short for better performance."
elif "memory" in error_message.lower():
return None, "Memory error: The model ran out of memory. Try with shorter text or a smaller voice sample."
else:
return None, f"Error generating speech: {error_message}"
# Create the Gradio interface
with gr.Blocks(theme=gr.themes.Soft()) as demo:
gr.Markdown("# XTTS Voice Cloning")
gr.Markdown("""
This app uses XTTS v2 to clone a voice from a sample and generate speech with that voice.
⚠️ **CPU Warning**: This is running on a CPU, so generation may take 1-3 minutes.
Keep your text short (1-2 sentences) for best results.
""")
with gr.Row():
with gr.Column():
text_input = gr.Textbox(
label="Text to speech",
placeholder="Enter text to be spoken in the cloned voice (keep it short)...",
lines=3
)
audio_input = gr.Audio(
label="Voice reference (5-10 second sample)",
type="filepath"
)
with gr.Row():
clear_btn = gr.Button("Clear")
submit_btn = gr.Button("Generate", variant="primary")
with gr.Column():
output_audio = gr.Audio(label="Generated Speech")
output_message = gr.Textbox(label="Status")
# Set up interactions
submit_btn.click(
fn=clone_voice,
inputs=[text_input, audio_input],
outputs=[output_audio, output_message]
)
clear_btn.click(
fn=lambda: (None, None, ""),
inputs=[],
outputs=[text_input, output_audio, output_message]
)
gr.Markdown("""
## Tips for better results:
1. Use a clear voice recording with minimal background noise
2. Keep the sample 5-10 seconds long
3. For faster generation, keep your text very short (1-2 sentences max)
4. Be patient! CPU generation takes time
## License
This app uses the XTTS v2 model under the Coqui Public Model License for non-commercial use only.
""")
# Comment out the problematic line - we don't need it
# demo.load(lambda: None, None, None, _js="() => {console.log('Loading model in the background...')}")
# Launch the app
demo.queue().launch()