""" Khmer/Multilingual TTS Dataset Generator — built on openbmb/VoxCPM2. Upload a JSON file of [{"id": ..., "text": ...}, ...] (up to 100 items), batch-generate speech for each entry, and download a zip containing: - audio/.wav for every item - metadata.csv and metadata.json (id, text, filename, status) Designed for preparing paired text/speech datasets for ASR & TTS training. """ import csv import json import os import tempfile import traceback import zipfile import gradio as gr import soundfile as sf MODEL = None MAX_ITEMS = 100 def load_model(): """Lazy-load VoxCPM2 once and cache it across requests.""" global MODEL if MODEL is None: from voxcpm import VoxCPM MODEL = VoxCPM.from_pretrained("openbmb/VoxCPM2", load_denoiser=False) return MODEL def parse_json_file(path): with open(path, "r", encoding="utf-8") as f: data = json.load(f) if not isinstance(data, list): raise ValueError("The JSON file must contain a list of objects, e.g. " '[{"id": 1, "text": "..."}, ...]') items = [] seen_ids = set() for i, entry in enumerate(data): if not isinstance(entry, dict) or "id" not in entry or "text" not in entry: raise ValueError(f"Item #{i} is missing 'id' or 'text': {entry}") item_id = entry["id"] text = str(entry["text"]).strip() if not text: raise ValueError(f"Item id={item_id} has empty text") if item_id in seen_ids: raise ValueError(f"Duplicate id found: {item_id}") seen_ids.add(item_id) items.append({"id": item_id, "text": text}) return items def build_zip(work_dir, audio_dir, metadata): csv_path = os.path.join(work_dir, "metadata.csv") with open(csv_path, "w", newline="", encoding="utf-8") as f: writer = csv.DictWriter(f, fieldnames=["id", "text", "filename", "status"]) writer.writeheader() writer.writerows(metadata) json_path = os.path.join(work_dir, "metadata.json") with open(json_path, "w", encoding="utf-8") as f: json.dump(metadata, f, ensure_ascii=False, indent=2) zip_path = os.path.join(work_dir, "tts_dataset.zip") with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zf: for fname in sorted(os.listdir(audio_dir)): zf.write(os.path.join(audio_dir, fname), arcname=f"audio/{fname}") zf.write(csv_path, arcname="metadata.csv") zf.write(json_path, arcname="metadata.json") return zip_path def generate_batch(json_file, voice_description, cfg_value, inference_timesteps, limit, progress=gr.Progress()): if json_file is None: raise gr.Error("Upload a JSON file first.") try: items = parse_json_file(json_file) except Exception as e: raise gr.Error(f"Could not read JSON: {e}") if len(items) == 0: raise gr.Error("The JSON file is empty.") if len(items) > MAX_ITEMS: raise gr.Error( f"Found {len(items)} items — please split into batches of " f"{MAX_ITEMS} or fewer." ) if limit and int(limit) > 0: items = items[: int(limit)] progress(0, desc="Loading VoxCPM2 (first run only, can take a while)...") model = load_model() sample_rate = getattr(model.tts_model, "sample_rate", 48000) work_dir = tempfile.mkdtemp(prefix="tts_batch_") audio_dir = os.path.join(work_dir, "audio") os.makedirs(audio_dir, exist_ok=True) metadata = [] rows = [] total = len(items) for i, item in enumerate(items): item_id = item["id"] text = item["text"] progress(i / total, desc=f"id={item_id} ({i + 1}/{total})") prompt_text = f"({voice_description.strip()}){text}" if voice_description.strip() else text status = "ok" filename = "" try: wav = model.generate( text=prompt_text, cfg_value=float(cfg_value), inference_timesteps=int(inference_timesteps), ) filename = f"{item_id}.wav" sf.write(os.path.join(audio_dir, filename), wav, sample_rate) except Exception as e: status = f"error: {e}" traceback.print_exc() metadata.append({"id": item_id, "text": text, "filename": filename, "status": status}) rows.append([item_id, text, filename, status]) yield rows, None, f"Processed {i + 1}/{total}" zip_path = build_zip(work_dir, audio_dir, metadata) ok_count = sum(1 for m in metadata if m["status"] == "ok") yield rows, zip_path, f"Done — {ok_count}/{total} generated successfully." with gr.Blocks(title="TTS Dataset Generator (VoxCPM2)") as demo: gr.Markdown( """ # 🗣️ Text → Speech Dataset Generator Built on [`openbmb/VoxCPM2`](https://huggingface.co/openbmb/VoxCPM2) (2B params, 30 languages including Khmer). Upload a JSON file shaped like `[{"id": 1, "text": "..."}, ...]` (up to 100 items) and generate a downloadable dataset: one `.wav` per item plus `metadata.csv` / `metadata.json` mapping id → text → filename — ready for ASR/TTS training. ⚠️ **This Space runs on free CPU hardware**, so generation is slow — expect roughly tens of seconds per sentence. Test with a small **limit** first before running a full batch of 100, and keep this tab open while it runs. """ ) with gr.Row(): with gr.Column(): json_input = gr.File(label="Upload JSON file", file_types=[".json"], type="filepath") voice_description = gr.Textbox( label="Voice description (optional, applied to every item)", placeholder="e.g. a calm young woman, clear and steady voice", ) with gr.Row(): cfg_value = gr.Slider(0.5, 4.0, value=2.0, step=0.1, label="CFG value") inference_timesteps = gr.Slider(4, 30, value=10, step=1, label="Inference timesteps") limit = gr.Number( label="Limit (0 = process all items — use a small number to test first)", value=5, precision=0, ) generate_btn = gr.Button("Generate batch", variant="primary") with gr.Column(): status_box = gr.Textbox(label="Status", interactive=False) results_table = gr.Dataframe( headers=["id", "text", "filename", "status"], label="Results", wrap=True, ) download_file = gr.File(label="Download dataset (.zip: audio/ + metadata.csv + metadata.json)") generate_btn.click( generate_batch, inputs=[json_input, voice_description, cfg_value, inference_timesteps, limit], outputs=[results_table, download_file, status_box], ) demo.queue(max_size=10).launch()