Spaces:
Runtime error
Runtime error
| # Generated by Copilot | |
| """ | |
| Persistent Storage Manager for TTS Project | |
| Utilizes 20GB permanent storage for saving outputs, models, and data | |
| """ | |
| import os | |
| import json | |
| import shutil | |
| from pathlib import Path | |
| from datetime import datetime | |
| from typing import Dict, List, Optional, Union | |
| import zipfile | |
| import tempfile | |
| from dataclasses import dataclass, asdict | |
| class StorageStats: | |
| """Storage statistics""" | |
| total_space: int | |
| used_space: int | |
| free_space: int | |
| num_files: int | |
| num_audio_files: int | |
| num_models: int | |
| class PersistentStorageManager: | |
| """Manages 20GB persistent storage for TTS project""" | |
| def __init__(self, base_path: str = "/data"): | |
| """Initialize storage manager with persistent storage path""" | |
| self.base_path = Path(base_path) | |
| self.ensure_directories() | |
| # Storage structure | |
| self.paths = { | |
| "audio_outputs": self.base_path / "audio_outputs", | |
| "batch_results": self.base_path / "batch_results", | |
| "voice_samples": self.base_path / "voice_samples", | |
| "models_cache": self.base_path / "models_cache", | |
| "user_data": self.base_path / "user_data", | |
| "analytics": self.base_path / "analytics", | |
| "webhooks_logs": self.base_path / "webhooks_logs", | |
| "exports": self.base_path / "exports", | |
| "backups": self.base_path / "backups" | |
| } | |
| def ensure_directories(self): | |
| """Create necessary directory structure""" | |
| directories = [ | |
| "audio_outputs", | |
| "batch_results", | |
| "voice_samples", | |
| "models_cache", | |
| "user_data", | |
| "analytics", | |
| "webhooks_logs", | |
| "exports", | |
| "backups" | |
| ] | |
| for directory in directories: | |
| dir_path = self.base_path / directory | |
| dir_path.mkdir(parents=True, exist_ok=True) | |
| # Create README files for each directory | |
| readme_path = dir_path / "README.md" | |
| if not readme_path.exists(): | |
| readme_content = f"""# {directory.replace('_', ' ').title()} | |
| This directory stores {directory.replace('_', ' ')} for the TTS project. | |
| - **Created**: {datetime.now().isoformat()} | |
| - **Purpose**: Persistent storage for TTS project data | |
| - **Storage**: Part of 20GB permanent storage allocation | |
| """ | |
| readme_path.write_text(readme_content) | |
| def save_audio_output(self, audio_path: str, metadata: Dict, user_id: str = "default") -> str: | |
| """Save audio output with metadata to persistent storage""" | |
| timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") | |
| filename = f"tts_{timestamp}_{user_id}.wav" | |
| # Create user directory | |
| user_dir = self.paths["audio_outputs"] / user_id | |
| user_dir.mkdir(exist_ok=True) | |
| # Save audio file | |
| dest_path = user_dir / filename | |
| shutil.copy2(audio_path, dest_path) | |
| # Save metadata | |
| metadata_path = user_dir / f"{filename}.json" | |
| metadata_with_info = { | |
| **metadata, | |
| "saved_at": datetime.now().isoformat(), | |
| "file_size": dest_path.stat().st_size, | |
| "original_path": audio_path | |
| } | |
| with open(metadata_path, 'w') as f: | |
| json.dump(metadata_with_info, f, indent=2) | |
| return str(dest_path) | |
| def save_batch_results(self, batch_files: List[str], batch_metadata: Dict) -> str: | |
| """Save batch processing results as ZIP with metadata""" | |
| timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") | |
| batch_name = f"batch_{timestamp}" | |
| # Create batch directory | |
| batch_dir = self.paths["batch_results"] / batch_name | |
| batch_dir.mkdir(exist_ok=True) | |
| # Copy files to batch directory | |
| saved_files = [] | |
| for i, file_path in enumerate(batch_files): | |
| if os.path.exists(file_path): | |
| dest_name = f"batch_{i:03d}.wav" | |
| dest_path = batch_dir / dest_name | |
| shutil.copy2(file_path, dest_path) | |
| saved_files.append(str(dest_path)) | |
| # Create ZIP archive | |
| zip_path = self.paths["exports"] / f"{batch_name}.zip" | |
| with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zipf: | |
| for file_path in saved_files: | |
| zipf.write(file_path, Path(file_path).name) | |
| # Save metadata | |
| metadata_path = batch_dir / "metadata.json" | |
| full_metadata = { | |
| **batch_metadata, | |
| "batch_id": batch_name, | |
| "created_at": datetime.now().isoformat(), | |
| "num_files": len(saved_files), | |
| "zip_path": str(zip_path), | |
| "files": saved_files | |
| } | |
| with open(metadata_path, 'w') as f: | |
| json.dump(full_metadata, f, indent=2) | |
| return str(zip_path) | |
| def save_voice_sample(self, audio_path: str, voice_name: str, metadata: Dict) -> str: | |
| """Save voice cloning reference samples""" | |
| voice_dir = self.paths["voice_samples"] / voice_name | |
| voice_dir.mkdir(exist_ok=True) | |
| timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") | |
| filename = f"{voice_name}_{timestamp}.wav" | |
| dest_path = voice_dir / filename | |
| shutil.copy2(audio_path, dest_path) | |
| # Save voice metadata | |
| voice_metadata = { | |
| **metadata, | |
| "voice_name": voice_name, | |
| "saved_at": datetime.now().isoformat(), | |
| "file_path": str(dest_path), | |
| "file_size": dest_path.stat().st_size | |
| } | |
| metadata_path = voice_dir / f"{filename}.json" | |
| with open(metadata_path, 'w') as f: | |
| json.dump(voice_metadata, f, indent=2) | |
| return str(dest_path) | |
| def cache_model(self, model_name: str, model_path: str) -> str: | |
| """Cache downloaded models for faster loading""" | |
| model_dir = self.paths["models_cache"] / model_name.replace("/", "_") | |
| model_dir.mkdir(exist_ok=True) | |
| if os.path.isdir(model_path): | |
| # Copy entire model directory | |
| dest_path = model_dir / "model" | |
| if dest_path.exists(): | |
| shutil.rmtree(dest_path) | |
| shutil.copytree(model_path, dest_path) | |
| else: | |
| # Copy single model file | |
| dest_path = model_dir / Path(model_path).name | |
| shutil.copy2(model_path, dest_path) | |
| # Save model info | |
| info_path = model_dir / "model_info.json" | |
| model_info = { | |
| "model_name": model_name, | |
| "cached_at": datetime.now().isoformat(), | |
| "original_path": model_path, | |
| "cached_path": str(dest_path), | |
| "size": self._get_directory_size(dest_path) if dest_path.is_dir() else dest_path.stat().st_size | |
| } | |
| with open(info_path, 'w') as f: | |
| json.dump(model_info, f, indent=2) | |
| return str(dest_path) | |
| def log_webhook_event(self, event_data: Dict) -> str: | |
| """Log webhook events to persistent storage""" | |
| date_str = datetime.now().strftime("%Y%m%d") | |
| log_file = self.paths["webhooks_logs"] / f"webhooks_{date_str}.jsonl" | |
| event_entry = { | |
| **event_data, | |
| "logged_at": datetime.now().isoformat() | |
| } | |
| with open(log_file, 'a') as f: | |
| f.write(json.dumps(event_entry) + '\n') | |
| return str(log_file) | |
| def save_analytics_data(self, analytics_data: Dict, data_type: str = "usage") -> str: | |
| """Save analytics data for long-term analysis""" | |
| date_str = datetime.now().strftime("%Y%m%d") | |
| analytics_file = self.paths["analytics"] / f"{data_type}_{date_str}.json" | |
| # Load existing data if file exists | |
| if analytics_file.exists(): | |
| with open(analytics_file, 'r') as f: | |
| existing_data = json.load(f) | |
| else: | |
| existing_data = {"entries": []} | |
| # Add new entry | |
| entry = { | |
| **analytics_data, | |
| "timestamp": datetime.now().isoformat() | |
| } | |
| existing_data["entries"].append(entry) | |
| # Save updated data | |
| with open(analytics_file, 'w') as f: | |
| json.dump(existing_data, f, indent=2) | |
| return str(analytics_file) | |
| def create_backup(self, backup_name: str = None) -> str: | |
| """Create backup of important data""" | |
| if backup_name is None: | |
| backup_name = f"backup_{datetime.now().strftime('%Y%m%d_%H%M%S')}" | |
| backup_path = self.paths["backups"] / f"{backup_name}.zip" | |
| with zipfile.ZipFile(backup_path, 'w', zipfile.ZIP_DEFLATED) as zipf: | |
| # Backup audio outputs (recent ones) | |
| for audio_file in self.paths["audio_outputs"].rglob("*.wav"): | |
| # Only backup files from last 30 days | |
| if (datetime.now().timestamp() - audio_file.stat().st_mtime) < (30 * 24 * 3600): | |
| arcname = str(audio_file.relative_to(self.base_path)) | |
| zipf.write(audio_file, arcname) | |
| # Backup voice samples | |
| for voice_file in self.paths["voice_samples"].rglob("*"): | |
| if voice_file.is_file(): | |
| arcname = str(voice_file.relative_to(self.base_path)) | |
| zipf.write(voice_file, arcname) | |
| # Backup analytics | |
| for analytics_file in self.paths["analytics"].rglob("*.json"): | |
| arcname = str(analytics_file.relative_to(self.base_path)) | |
| zipf.write(analytics_file, arcname) | |
| return str(backup_path) | |
| def get_storage_stats(self) -> StorageStats: | |
| """Get storage usage statistics""" | |
| total_size = 20 * 1024 * 1024 * 1024 # 20GB in bytes | |
| used_size = self._get_directory_size(self.base_path) | |
| # Count files | |
| audio_files = len(list(self.paths["audio_outputs"].rglob("*.wav"))) | |
| total_files = len(list(self.base_path.rglob("*"))) | |
| model_dirs = len(list(self.paths["models_cache"].iterdir())) | |
| return StorageStats( | |
| total_space=total_size, | |
| used_space=used_size, | |
| free_space=total_size - used_size, | |
| num_files=total_files, | |
| num_audio_files=audio_files, | |
| num_models=model_dirs | |
| ) | |
| def cleanup_old_files(self, days: int = 30): | |
| """Clean up files older than specified days""" | |
| cutoff_time = datetime.now().timestamp() - (days * 24 * 3600) | |
| cleaned_files = [] | |
| for file_path in self.base_path.rglob("*"): | |
| if file_path.is_file() and file_path.stat().st_mtime < cutoff_time: | |
| # Don't delete model cache or voice samples | |
| if "models_cache" not in str(file_path) and "voice_samples" not in str(file_path): | |
| file_path.unlink() | |
| cleaned_files.append(str(file_path)) | |
| return cleaned_files | |
| def _get_directory_size(self, directory: Path) -> int: | |
| """Get total size of directory""" | |
| total_size = 0 | |
| for file_path in directory.rglob("*"): | |
| if file_path.is_file(): | |
| total_size += file_path.stat().st_size | |
| return total_size | |
| def list_saved_outputs(self, user_id: str = None, limit: int = 50) -> List[Dict]: | |
| """List saved audio outputs with metadata""" | |
| outputs = [] | |
| search_path = self.paths["audio_outputs"] | |
| if user_id: | |
| search_path = search_path / user_id | |
| if not search_path.exists(): | |
| return outputs | |
| # Find audio files and their metadata | |
| for audio_file in search_path.rglob("*.wav"): | |
| metadata_file = audio_file.with_suffix(".wav.json") | |
| if metadata_file.exists(): | |
| try: | |
| with open(metadata_file, 'r') as f: | |
| metadata = json.load(f) | |
| outputs.append({ | |
| "file_path": str(audio_file), | |
| "metadata": metadata, | |
| "size": audio_file.stat().st_size, | |
| "created": datetime.fromtimestamp(audio_file.stat().st_ctime).isoformat() | |
| }) | |
| except Exception as e: | |
| print(f"Error reading metadata for {audio_file}: {e}") | |
| # Sort by creation time (newest first) and limit results | |
| outputs.sort(key=lambda x: x["created"], reverse=True) | |
| return outputs[:limit] | |
| # Global storage manager instance | |
| storage_manager = PersistentStorageManager() |