anycoder-0cd6b60f / utils.py
Asakrg's picture
Upload utils.py with huggingface_hub
3f1dadd verified
Raw History Blame Contribute Delete
2.13 kB
"""
Utility functions for VibeVoice TTS Tester
"""
import os
import tempfile
import base64
from typing import Optional, Tuple
def format_duration(seconds: float) -> str:
"""Format duration in seconds to human-readable string."""
if seconds < 60:
return f"{seconds:.1f}s"
minutes = int(seconds // 60)
secs = seconds % 60
return f"{minutes}m {secs:.1f}s"
def get_audio_duration(audio_path: str) -> Optional[float]:
"""Get the duration of an audio file in seconds."""
try:
import wave
with wave.open(audio_path, 'rb') as audio_file:
frames = audio_file.getnframes()
rate = audio_file.getframerate()
duration = frames / float(rate)
return duration
except Exception:
return None
def create_temp_audio_file(audio_bytes: bytes, suffix: str = ".wav") -> str:
"""Create a temporary audio file and return its path."""
with tempfile.NamedTemporaryFile(delete=False, suffix=suffix) as tmp_file:
tmp_file.write(audio_bytes)
return tmp_file.name
def audio_to_base64(audio_path: str) -> str:
"""Convert audio file to base64 string for embedding."""
with open(audio_path, 'rb') as audio_file:
audio_bytes = audio_file.read()
return base64.b64encode(audio_bytes).decode('utf-8')
def validate_text_input(text: str, max_length: int = 5000) -> Tuple[bool, str]:
"""Validate text input for TTS."""
if not text or not text.strip():
return False, "Text cannot be empty."
if len(text) > max_length:
return False, f"Text exceeds maximum length of {max_length} characters."
return True, "Valid input."
def estimate_generation_time(text: str) -> float:
"""Estimate the generation time based on text length."""
base_time = 2.0
char_time = len(text) * 0.05
return base_time + char_time
def cleanup_temp_files(file_paths: list) -> None:
"""Clean up temporary files."""
for path in file_paths:
try:
if os.path.exists(path):
os.remove(path)
except Exception:
pass