Chatbot / app /tts.py
ShadowTEM's picture
removed the main
7ac556e verified
Raw
History Blame Contribute Delete
5.11 kB
import builtins
import os
import asyncio
import numpy as np
import time
import json
import soundfile as sf
from kokoro_onnx import Kokoro, SAMPLE_RATE
class tts():
# Update default voice_path to the JSON file as recommended.
def __init__(self, tts_model_path="/app/models/kokoro-v1.0.fp16.onnx", voice_path="/app/models/voices-v1.0.bin"):
self.tts_model_path = tts_model_path
self.voice_path = voice_path
self.tts = None
self.initialize_tts()
try:
import nest_asyncio
nest_asyncio.apply()
except ImportError:
pass
def initialize_tts(self):
# Check that the voices file exists.
if not os.path.exists(self.voice_path):
raise FileNotFoundError(
f"Voices file not found at {self.voice_path}\n"
"You can download the voices file using the following command:\n"
"wget https://github.com/thewh1teagle/kokoro-onnx/releases/download/model-files/voices.json"
)
# Load the TTS model and configuration.
self.tts = Kokoro(self.tts_model_path, self.voice_path)
print("TTS initialized.")
def insert_chunks_into_text(self, text, tokens_per_chunk=24):
tokens = text.split()
chunks = []
formatted_text = ""
chunk_count = 0
current_chunk = []
for token in tokens:
current_chunk.append(token)
if len(current_chunk) == tokens_per_chunk:
chunk_count += 1
chunk_str = " ".join(current_chunk)
chunks.append(chunk_str)
formatted_text += f"[Chunk {chunk_count}]: {chunk_str}\n\n"
current_chunk = []
if current_chunk:
chunk_count += 1
chunk_str = " ".join(current_chunk)
chunks.append(chunk_str)
formatted_text += f"[Chunk {chunk_count}]: {chunk_str}\n\n"
return {
"original_text": text,
"formatted_text": formatted_text,
"chunks": chunks
}
async def cpu_stream_to_audio(self, text, voice="af_heart", speed=1.2, lang="en-us"):
stream = self.tts.create_stream(text, voice, speed=speed, lang=lang)
audio_chunks = []
async for samples, sample_rate in stream:
audio_chunks.append(samples)
if not audio_chunks:
raise RuntimeError("No audio chunks were received from the stream.")
audio_data = np.concatenate(audio_chunks, axis=0)
return audio_data, sample_rate
# if __name__ == "__main__":
# tts_instance = tts()
# lyrics_lines = [
# "Ooh",
# "I, I just woke up from a dream",
# "Where you and I had to say goodbye",
# "And I don't know what it all means",
# "But since I survived, I realized",
# "Wherever you go, that's where I'll follow",
# "Nobody's promised tomorrow",
# "So I'ma love you every night like it's the last night",
# "Like it's the last night",
# "If the world was ending, I'd wanna be next to you",
# "If the party was over and our time on Earth was through",
# "I'd wanna hold you just for a while and die with a smile",
# "If the world was ending, I'd wanna be next to you",
# "Ooh",
# "Ooh, lost, lost in the words that we scream",
# "I don't even wanna do this anymore",
# "'Cause you already know what you mean to me",
# "And our love's the only war worth fighting for",
# "Wherever you go, that's where I'll follow",
# "Nobody's promised tomorrow",
# "So I'ma love you every night like it's the last night",
# "Like it's the last night",
# "If the world was ending, I'd wanna be next to you",
# "If the party was over and our time on Earth was through",
# "I'd wanna hold you just for a while and die with a smile",
# "If the world was ending, I'd wanna be next to you",
# "Right next to you",
# "Next to you",
# "Right next to you",
# "Oh-oh, oh",
# "If the world was ending, I'd wanna be next to you",
# "If the party was over and our time on Earth was through",
# "I'd wanna hold you just for a while and die with a smile",
# "If the world was ending, I'd wanna be next to you",
# "If the world was ending, I'd wanna be next to you",
# "Ooh",
# "I'd wanna be next to you"
# ]
# lyrics_text = "\n".join(lyrics_lines)
# chunked_text = tts_instance.insert_chunks_into_text(lyrics_text)
# start = time.time()
# audio_data, sample_rate = asyncio.run(tts_instance.cpu_stream_to_audio(chunked_text['original_text']))
# print(f"Transcription complete in {time.time() - start:.2f} seconds")
# print(f"Returned audio data shape: {audio_data.shape} with sample rate: {sample_rate}")
# output_filename = "output.wav"
# sf.write(output_filename, audio_data, sample_rate)
# print(f"Audio saved as {output_filename}")