aiseosae commited on 10 days ago

Commit

87c19a6

verified ·

1 Parent(s): 758e465

Upload folder using huggingface_hub

Browse files

Files changed (19) hide show

.gitattributes +1 -0
README.md +104 -0
added_tokens.json +35 -0
chute_config.yml +23 -0
config.json +163 -0
generation_config.json +12 -0
merges.txt +0 -0
miner.py +208 -0
model.safetensors +3 -0
preprocessor_config.json +6 -0
special_tokens_map.json +44 -0
speech_tokenizer/config.json +94 -0
speech_tokenizer/configuration.json +1 -0
speech_tokenizer/model.safetensors +3 -0
speech_tokenizer/preprocessor_config.json +10 -0
tokenizer.json +3 -0
tokenizer_config.json +318 -0
vocab.json +0 -0
vocence_config.yaml +16 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

README.md ADDED Viewed

	@@ -0,0 +1,104 @@

+---
+license: cc-by-nc-sa-4.0
+base_model: Qwen/Qwen3-TTS-12Hz-1.7B-VoiceDesign
+pipeline_tag: text-to-speech
+library_name: transformers
+language:
+  - en
+tags:
+  - tts
+  - qwen3-tts
+  - voice-design
+  - prompttts
+  - vocence
+  - bittensor
+---
+Inference uses **`qwen_tts.Qwen3TTSModel`**, loaded from the repo root via `from_pretrained(this_folder)`.
+## Layout
+| Path | Role |
+|------|------|
+| `config.json`, weights, tokenizer, codec dirs | Qwen3-TTS snapshot (as shipped by the upstream model card) |
+| `miner.py` | Vocence engine: `Miner`, `warmup()`, `generate_wav(instruction, text)` |
+| `vocence_config.yaml` | Device, dtype, caps, language |
+| `chute_config.yml` | Chutes image / GPU / scaling / TEE |
+| `demo.py` | Optional local smoke test (if present) |
+## Vocence API
+Validators call your deployed chute with JSON shaped like:
+```json
+{
+  "text": "Words to speak.",
+  "instruction": "gender: male | pitch: mid | speed: normal | age_group: adult | emotion: neutral | tone: casual | accent: us"
+}
+```
+The miner forwards **`text`** → `generate_voice_design(..., text=...)` and **`instruction`** → `instruct=...`, using **`language`** from config (default English).
+## Configure (`vocence_config.yaml`)
+| Area | Keys |
+|------|------|
+| Runtime | `device_preference` (`cuda` / `cpu`), `dtype` (`bfloat16` / `float32`), `use_flash_attention_2`, `default_language` |
+| Generation | `sample_rate` (e.g. 24000), `max_seconds` |
+| Limits | `max_text_chars`, `max_instruction_chars`, `default_language` |
+Warmup runs one short `generate_voice_design` with a **180 s** timeout.
+## Local quick test
+Install PyTorch (CUDA if available), then:
+```bash
+pip install "qwen-tts" pyyaml soundfile numpy
+```
+```python
+from pathlib import Path
+from miner import Miner
+miner = Miner(Path("."))
+miner.warmup()
+wave, sr = miner.generate_wav(
+    instruction="A calm, clear narrator, neutral US accent.",
+    text="Hello — this is a short synthesis check.",
+)
+```
+Or load the class directly from transformers-style layout:
+```python
+from qwen_tts import Qwen3TTSModel
+model = Qwen3TTSModel.from_pretrained(".")  # or your HF repo id
+wavs, sr = model.generate_voice_design(
+    text="Hello fellas.",
+    instruct="Cute voice.",
+    language="english",
+)
+```
+Replace `"."` with your HF repo id after upload, e.g. `"your-org/your-repo"`.
+## Chutes / Vocence deploy
+1. Push this layout to a Hugging Face **model** repo; pin a **commit SHA** for `VOCENCE_REVISION`.
+2. Render the canonical Vocence chute script with `VOCENCE_REPO`, `VOCENCE_REVISION`, `VOCENCE_CHUTES_USER`, `VOCENCE_CHUTE_ID`.
+3. `chutes build … --wait` then `chutes deploy … --accept-fee`.
+4. Commit on chain: `model_name`, `model_revision` (HF SHA), `chute_id` (UUID from Chutes).
+Chute **name** must contain **`vocence`** (case-insensitive). See **`miner_sample/MINER_GUIDE.md`** in the Vocence repo.
+## Training / fine-tuning
+Fine-tuning is done **outside** Chutes on your own GPU; export a full snapshot compatible with **`Qwen3TTSModel.from_pretrained(...)`**, then replace weights in this repo layout and push a new revision.
+## License
+**CC BY-NC-SA 4.0** — see the license file in this repo. Respect upstream Qwen / Alibaba terms for the base checkpoint.

added_tokens.json ADDED Viewed

	@@ -0,0 +1,35 @@

+{
+  "</think>": 151668,
+  "</tool_call>": 151658,
+  "</tool_response>": 151666,
+  "<think>": 151667,
+  "<tool_call>": 151657,
+  "<tool_response>": 151665,
+  "<tts_pad>": 151671,
+  "<tts_text_bos>": 151672,
+  "<tts_text_bos_single>": 151674,
+  "<tts_text_eod>": 151673,
+  "<|audio_end|>": 151670,
+  "<|audio_pad|>": 151675,
+  "<|audio_start|>": 151669,
+  "<|box_end|>": 151649,
+  "<|box_start|>": 151648,
+  "<|endoftext|>": 151643,
+  "<|file_sep|>": 151664,
+  "<|fim_middle|>": 151660,
+  "<|fim_pad|>": 151662,
+  "<|fim_prefix|>": 151659,
+  "<|fim_suffix|>": 151661,
+  "<|im_end|>": 151645,
+  "<|im_start|>": 151644,
+  "<|image_pad|>": 151655,
+  "<|object_ref_end|>": 151647,
+  "<|object_ref_start|>": 151646,
+  "<|quad_end|>": 151651,
+  "<|quad_start|>": 151650,
+  "<|repo_name|>": 151663,
+  "<|video_pad|>": 151656,
+  "<|vision_end|>": 151653,
+  "<|vision_pad|>": 151654,
+  "<|vision_start|>": 151652
+}

chute_config.yml ADDED Viewed

	@@ -0,0 +1,23 @@

+# Image + node + Chute for Vocence deploy. Required in the HF repo at build time.
+Image:
+  from_base: parachutes/python:3.12
+  run_command:
+    - pip install torch torchaudio transformers accelerate huggingface_hub pyyaml soundfile librosa
+    - pip install -U qwen-tts
+  set_workdir: /app
+NodeSelector:
+  gpu_count: 1
+  min_vram_gb_per_gpu: 24
+  include: ["pro_6000"]
+  exclude: []
+Chute:
+  tagline: Vocence TTS — Qwen3 PromptTTS (weights in repo)
+  readme: Qwen3 12Hz TTS snapshot + miner.py for Vocence
+  shutdown_after_seconds: 86400
+  concurrency: 1
+  max_instances: 2
+  scaling_threshold: 0.5
+  tee: true

config.json ADDED Viewed

	@@ -0,0 +1,163 @@

+{
+  "architectures": [
+    "Qwen3TTSForConditionalGeneration"
+  ],
+  "assistant_token_id": 77091,
+  "im_end_token_id": 151645,
+  "im_start_token_id": 151644,
+  "tts_bos_token_id": 151672,
+  "tts_eos_token_id": 151673,
+  "tts_pad_token_id": 151671,
+  "model_type": "qwen3_tts",
+  "tokenizer_type": "qwen3_tts_tokenizer_12hz",
+  "tts_model_size": "1b7",
+  "tts_model_type": "voice_design",
+  "talker_config": {
+    "attention_bias": false,
+    "attention_dropout": 0,
+    "code_predictor_config": {
+      "_name_or_path": "",
+      "add_cross_attention": false,
+      "architectures": null,
+      "attention_bias": false,
+      "attention_dropout": 0,
+      "bad_words_ids": null,
+      "begin_suppress_tokens": null,
+      "bos_token_id": null,
+      "chunk_size_feed_forward": 0,
+      "cross_attention_hidden_size": null,
+      "decoder_start_token_id": null,
+      "diversity_penalty": 0.0,
+      "do_sample": false,
+      "early_stopping": false,
+      "encoder_no_repeat_ngram_size": 0,
+      "eos_token_id": null,
+      "exponential_decay_length_penalty": null,
+      "finetuning_task": null,
+      "forced_bos_token_id": null,
+      "forced_eos_token_id": null,
+      "head_dim": 128,
+      "hidden_act": "silu",
+      "hidden_size": 1024,
+      "id2label": {
+        "0": "LABEL_0",
+        "1": "LABEL_1"
+      },
+      "initializer_range": 0.02,
+      "intermediate_size": 3072,
+      "is_decoder": false,
+      "is_encoder_decoder": false,
+      "label2id": {
+        "LABEL_0": 0,
+        "LABEL_1": 1
+      },
+      "layer_types": [
+        "full_attention",
+        "full_attention",
+        "full_attention",
+        "full_attention",
+        "full_attention"
+      ],
+      "length_penalty": 1.0,
+      "max_length": 20,
+      "max_position_embeddings": 65536,
+      "max_window_layers": 28,
+      "min_length": 0,
+      "model_type": "qwen3_tts_talker_code_predictor",
+      "no_repeat_ngram_size": 0,
+      "num_attention_heads": 16,
+      "num_beam_groups": 1,
+      "num_beams": 1,
+      "num_code_groups": 16,
+      "num_hidden_layers": 5,
+      "num_key_value_heads": 8,
+      "num_return_sequences": 1,
+      "output_attentions": false,
+      "output_hidden_states": false,
+      "output_scores": false,
+      "pad_token_id": null,
+      "prefix": null,
+      "problem_type": null,
+      "pruned_heads": {},
+      "remove_invalid_values": false,
+      "repetition_penalty": 1.0,
+      "return_dict": true,
+      "return_dict_in_generate": false,
+      "rms_norm_eps": 1e-06,
+      "rope_scaling": null,
+      "rope_theta": 1000000,
+      "sep_token_id": null,
+      "sliding_window": null,
+      "suppress_tokens": null,
+      "task_specific_params": null,
+      "temperature": 1.0,
+      "tf_legacy_loss": false,
+      "tie_encoder_decoder": false,
+      "tie_word_embeddings": false,
+      "tokenizer_class": null,
+      "top_k": 50,
+      "top_p": 1.0,
+      "dtype": null,
+      "torchscript": false,
+      "typical_p": 1.0,
+      "use_bfloat16": false,
+      "use_cache": true,
+      "use_sliding_window": false,
+      "vocab_size": 2048
+    },
+    "codec_bos_id": 2149,
+    "codec_eos_token_id": 2150,
+    "codec_think_id": 2154,
+    "codec_language_id": {
+        "chinese": 2055,
+        "english": 2050,
+        "german": 2053,
+        "italian": 2070,
+        "portuguese": 2071,
+        "spanish": 2054,
+        "japanese": 2058,
+        "korean": 2064,
+        "french": 2061,
+        "russian": 2069
+    },
+    "codec_nothink_id": 2155,
+    "codec_pad_id": 2148,
+    "codec_think_bos_id": 2156,
+    "codec_think_eos_id": 2157,
+    "spk_id": {
+    },
+    "spk_is_dialect": {
+    },
+    "head_dim": 128,
+    "hidden_act": "silu",
+    "hidden_size": 2048,
+    "initializer_range": 0.02,
+    "intermediate_size": 6144,
+    "max_position_embeddings": 32768,
+    "model_type": "qwen3_tts_talker",
+    "num_attention_heads": 16,
+    "num_code_groups": 16,
+    "num_hidden_layers": 28,
+    "num_key_value_heads": 8,
+    "position_id_per_seconds": 13,
+    "rms_norm_eps": 1e-06,
+    "rope_scaling": {
+      "interleaved": true,
+      "mrope_section": [
+        24,
+        20,
+        20
+      ],
+      "rope_type": "default",
+      "type": "default"
+    },
+    "rope_theta": 1000000,
+    "sliding_window": null,
+    "text_hidden_size": 2048,
+    "text_vocab_size": 151936,
+    "use_cache": true,
+    "use_sliding_window": false,
+    "vocab_size": 3072
+  },
+  "transformers_version": "4.57.3"
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,12 @@

+{
+  "do_sample": true,
+  "repetition_penalty": 1.05,
+  "temperature": 0.9,
+  "top_p": 1.0,
+  "top_k": 50,
+  "subtalker_dosample": true,
+  "subtalker_temperature": 0.9,
+  "subtalker_top_p": 1.0,
+  "subtalker_top_k": 50,
+  "max_new_tokens": 8192
+}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

miner.py ADDED Viewed

	@@ -0,0 +1,208 @@

+from __future__ import annotations
+from concurrent.futures import ThreadPoolExecutor, TimeoutError as FutureTimeout
+from pathlib import Path
+from typing import Any
+import numpy as np
+VOCENCE_CONFIG = "vocence_config.yaml"
+QWEN_ANCHOR = "config.json"
+WARMUP_SECONDS = 180.0
+# Vocence validators split prompts into speakable `text` and trait `instruction`.
+# Scoring weights script heavily (WER vs transcript) and compares extracted voice
+# traits to these keys — enrich instructions so Qwen voice-design follows them.
+def _trait_phrase(key: str, val: str) -> str | None:
+    k, v = key.strip().lower(), val.strip().lower()
+    if not v:
+        return None
+    if k == "gender":
+        return {
+            "male": "male speaker",
+            "female": "female speaker",
+            "neutral": "gender-neutral voice",
+        }.get(v, f"{v} gender presentation")
+    if k == "pitch":
+        return {"low": "low pitch", "mid": "mid pitch", "high": "high pitch"}.get(v, f"{v} pitch")
+    if k == "speed":
+        return {"slow": "slow pacing", "normal": "normal pacing", "fast": "fast pacing"}.get(
+            v, f"{v} speed"
+        )
+    if k == "age_group":
+        return {
+            "child": "childlike voice",
+            "young_adult": "young adult voice",
+            "adult": "adult voice",
+            "senior": "older adult voice",
+        }.get(v, f"{v} age character")
+    if k == "emotion":
+        return f"{v} emotional tone"
+    if k == "tone":
+        return f"{v} speaking tone"
+    if k == "accent":
+        return {
+            "us": "American English accent",
+            "uk": "British English accent",
+            "au": "Australian English accent",
+            "in": "Indian English accent",
+            "neutral": "neutral accent",
+            "other": "clear intelligible accent",
+        }.get(v, f"{v} accent")
+    return f"{k.replace('_', ' ')}: {v}"
+def _expand_vocence_instruction(raw: str) -> str:
+    """Turn `gender: x | pitch: y | ...` into fluent voice-design text for Qwen3."""
+    s = (raw or "").strip()
+    if not s:
+        return "Neutral, clear, natural speech."
+    # Vocence trait lines use pipes and key: value pairs (see README).
+    if "|" in s and ":" in s:
+        phrases: list[str] = []
+        for chunk in s.split("|"):
+            chunk = chunk.strip()
+            if ":" in chunk:
+                key, _, val = chunk.partition(":")
+                p = _trait_phrase(key, val)
+                if p:
+                    phrases.append(p)
+            elif chunk:
+                phrases.append(chunk)
+        if phrases:
+            return (
+                "Voice design — "
+                + "; ".join(phrases)
+                + ". Deliver the script naturally with clear articulation and human-like prosody."
+            )
+    return s
+def _load_yaml(path: Path) -> dict[str, Any]:
+    if not path.is_file():
+        return {}
+    from yaml import safe_load
+    with path.open("r", encoding="utf-8") as fh:
+        return safe_load(fh) or {}
+def _select_device(prefer_cuda: bool):
+    import torch
+    has_cuda = torch.cuda.is_available()
+    device = "cuda:0" if (prefer_cuda and has_cuda) else "cpu"
+    return device, torch, has_cuda
+def _select_dtype(torch_mod, want_bf16: bool, has_cuda: bool):
+    return torch_mod.bfloat16 if (want_bf16 and has_cuda) else torch_mod.float32
+def _build_qwen(snapshot: Path, device: str, dtype: Any, attn: str):
+    from qwen_tts import Qwen3TTSModel
+    return Qwen3TTSModel.from_pretrained(
+        pretrained_model_name_or_path=str(snapshot),
+        device_map=device,
+        dtype=dtype,
+        attn_implementation=attn,
+    )
+def _attn_order(prefer_flash: bool) -> tuple[str, ...]:
+    return ("flash_attention_2", "sdpa") if prefer_flash else ("sdpa",)
+def _mono_pcm(arr: Any) -> np.ndarray:
+    wave = np.asarray(arr, dtype=np.float32)
+    return wave.mean(axis=1) if wave.ndim > 1 else wave
+def _settings(snapshot: Path) -> dict[str, Any]:
+    raw = _load_yaml(snapshot / VOCENCE_CONFIG)
+    rt = raw.get("runtime") or {}
+    gen = raw.get("generation") or {}
+    lim = raw.get("limits") or {}
+    return {
+        "language": str(lim.get("default_language") or rt.get("default_language") or "English"),
+        "sample_rate": int(gen.get("sample_rate", 24000)),
+        "cap_instruct": int(lim.get("max_instruction_chars", 600)),
+        "cap_text": int(lim.get("max_text_chars", 2000)),
+        "prefer_cuda": str(rt.get("device_preference", "cuda")).lower() == "cuda",
+        "prefer_bf16": str(rt.get("dtype", "bfloat16")).lower() == "bfloat16",
+        "prefer_flash": bool(rt.get("use_flash_attention_2", False)),
+        # If True, rewrite trait pipes into richer English for the instruct field.
+        "expand_instruction": bool(rt.get("expand_vocence_instruction", True)),
+    }
+class Miner:
+    def __init__(self, path_hf_repo: Path) -> None:
+        snapshot = Path(path_hf_repo).resolve()
+        if not (snapshot / QWEN_ANCHOR).is_file():
+            raise FileNotFoundError(f"snapshot missing {QWEN_ANCHOR}: {snapshot}")
+        self.snapshot = snapshot
+        self.cfg = _settings(snapshot)
+        device, torch_mod, has_cuda = _select_device(self.cfg["prefer_cuda"])
+        dtype = _select_dtype(torch_mod, self.cfg["prefer_bf16"], has_cuda)
+        last_err: BaseException | None = None
+        engine = None
+        for attn in _attn_order(self.cfg["prefer_flash"]):
+            try:
+                engine = _build_qwen(snapshot, device, dtype, attn)
+                tag = "bf16" if self.cfg["prefer_bf16"] and has_cuda else "fp32"
+                print(f"[Miner] qwen3-tts ready: device={device} dtype={tag} attn={attn}")
+                break
+            except Exception as exc:
+                last_err = exc
+        if engine is None:
+            raise RuntimeError(f"qwen3-tts load failed: {last_err!r}")
+        self.engine = engine
+    def __repr__(self) -> str:
+        return f"<Miner snapshot={self.snapshot.name} lang={self.cfg['language']!r}>"
+    def warmup(self) -> None:
+        instruct = (
+            "Voice design — male speaker; mid pitch; normal pacing; adult voice; "
+            "neutral emotional tone; casual speaking tone; American English accent. "
+            "Deliver naturally with clear articulation."
+        )
+        with ThreadPoolExecutor(max_workers=1) as pool:
+            future = pool.submit(self.generate_wav, instruct, "Warmup phrase for inference.")
+            try:
+                future.result(timeout=WARMUP_SECONDS)
+            except FutureTimeout:
+                raise RuntimeError(f"Miner warmup exceeded {WARMUP_SECONDS}s")
+    def generate_wav(self, instruction: str, text: str) -> tuple[np.ndarray, int]:
+        """Synthesize mono float32 PCM.
+        **Do not rewrite `text`** — Vocence scores script via word error rate vs the
+        requested transcript. Only `instruction` guides voice traits and naturalness.
+        """
+        cap_i = self.cfg["cap_instruct"]
+        cap_t = self.cfg["cap_text"]
+        raw_instr = instruction[:cap_i] if cap_i > 0 else instruction
+        if self.cfg.get("expand_instruction", True):
+            expanded = _expand_vocence_instruction(raw_instr)
+            prompt = expanded[:cap_i] if cap_i > 0 else expanded
+        else:
+            prompt = raw_instr
+        body = text[:cap_t] if cap_t > 0 else text
+        wavs, sr = self.engine.generate_voice_design(
+            text=body,
+            instruct=prompt,
+            language=self.cfg["language"],
+        )
+        if not wavs or wavs[0] is None:
+            raise ValueError("qwen3-tts returned no audio")
+        return _mono_pcm(wavs[0]), int(sr)

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e4d4433151c6b6dd291818a3c7116cea1b2082542fbb70ce22e5ffb937c488fe
+size 3833402644

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "padding_side": "left",
+  "padding_value": 0.0,
+  "processor_class": "Qwen3TTSProcessor",
+  "return_attention_mask": true
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>",
+    "<|audio_start|>",
+    "<|audio_end|>",
+    "<tts_pad>",
+    "<tts_text_bos>",
+    "<tts_text_bos_single>",
+    "<|audio_pad|>"
+  ],
+  "audio_bos_token": "<|audio_start|>",
+  "audio_eos_token": "<|audio_end|>",
+  "audio_token": "<|audio_pad|>",
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "image_token": "<|image_pad|>",
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "video_token": "<|video_pad|>",
+  "vision_bos_token": "<|vision_start|>",
+  "vision_eos_token": "<|vision_end|>"
+}

speech_tokenizer/config.json ADDED Viewed

	@@ -0,0 +1,94 @@

+{
+  "architectures": [
+    "Qwen3TTSTokenizerV2Model"
+  ],
+  "model_type": "qwen3_tts_tokenizer_12hz",
+  "encoder_valid_num_quantizers": 16,
+  "input_sample_rate": 24000,
+  "output_sample_rate": 24000,
+  "decode_upsample_rate": 1920,
+  "encode_downsample_rate": 1920,
+  "decoder_config": {
+    "attention_bias": false,
+    "attention_dropout": 0.0,
+    "latent_dim": 1024,
+    "codebook_dim": 512,
+    "codebook_size": 2048,
+    "decoder_dim": 1536,
+    "hidden_act": "silu",
+    "hidden_size": 512,
+    "intermediate_size": 1024,
+    "layer_scale_initial_scale": 0.01,
+    "max_position_embeddings": 8000,
+    "head_dim": 64,
+    "num_attention_heads": 16,
+    "num_hidden_layers": 8,
+    "num_key_value_heads": 16,
+    "num_quantizers": 16,
+    "num_semantic_quantizers": 1,
+    "rms_norm_eps": 1e-05,
+    "rope_theta": 10000,
+    "semantic_codebook_size": 4096,
+    "sliding_window": 72,
+    "upsample_rates": [
+      8,
+      5,
+      4,
+      3
+    ],
+    "upsampling_ratios": [
+      2,
+      2
+    ],
+    "vector_quantization_hidden_dimension": 512
+  },
+  "encoder_config": {
+    "_frame_rate": 12.5,
+    "attention_bias": false,
+    "attention_dropout": 0.0,
+    "audio_channels": 1,
+    "codebook_dim": 256,
+    "codebook_size": 2048,
+    "compress": 2,
+    "dilation_growth_rate": 2,
+    "dtype": "float32",
+    "head_dim": 64,
+    "hidden_act": "gelu",
+    "hidden_size": 512,
+    "initializer_range": 0.02,
+    "intermediate_size": 2048,
+    "kernel_size": 7,
+    "last_kernel_size": 3,
+    "layer_scale_initial_scale": 0.01,
+    "max_position_embeddings": 8000,
+    "norm_eps": 1e-05,
+    "normalize": false,
+    "num_attention_heads": 8,
+    "num_filters": 64,
+    "num_hidden_layers": 8,
+    "num_key_value_heads": 8,
+    "num_quantizers": 32,
+    "num_residual_layers": 1,
+    "num_semantic_quantizers": 1,
+    "pad_mode": "constant",
+    "residual_kernel_size": 3,
+    "rope_theta": 10000.0,
+    "sampling_rate": 24000,
+    "sliding_window": 250,
+    "transformers_version": "4.57.0.dev0",
+    "trim_right_ratio": 1.0,
+    "upsample_groups": 512,
+    "upsampling_ratios": [
+      8,
+      6,
+      5,
+      4
+    ],
+    "use_cache": false,
+    "use_causal_conv": true,
+    "use_conv_shortcut": false,
+    "use_streaming": false,
+    "vector_quantization_hidden_dimension": 256
+  },
+  "transformers_version": "4.57.3"
+}

speech_tokenizer/configuration.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"framework": "pytorch", "task": "feature-extraction", "allow_remote": true}

speech_tokenizer/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:836b7b357f5ea43e889936a3709af68dfe3751881acefe4ecf0dbd30ba571258
+size 682293092

speech_tokenizer/preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+  "chunk_length_s": null,
+  "feature_extractor_type": "EncodecFeatureExtractor",
+  "feature_size": 1,
+  "overlap": null,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "return_attention_mask": true,
+  "sampling_rate": 24000
+}

tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:09267689b8362020b9763b65dd5be7e086b31e28d72e02837a9e781de9a91bc7
+size 11423986

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,318 @@

+{
+  "add_bos_token": false,
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "151643": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151644": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151645": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151646": {
+      "content": "<|object_ref_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151647": {
+      "content": "<|object_ref_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151648": {
+      "content": "<|box_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151649": {
+      "content": "<|box_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151650": {
+      "content": "<|quad_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151651": {
+      "content": "<|quad_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151652": {
+      "content": "<|vision_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151653": {
+      "content": "<|vision_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151654": {
+      "content": "<|vision_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151655": {
+      "content": "<|image_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151656": {
+      "content": "<|video_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151657": {
+      "content": "<tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151658": {
+      "content": "</tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151659": {
+      "content": "<|fim_prefix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151660": {
+      "content": "<|fim_middle|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151661": {
+      "content": "<|fim_suffix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151662": {
+      "content": "<|fim_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151663": {
+      "content": "<|repo_name|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151664": {
+      "content": "<|file_sep|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151665": {
+      "content": "<tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151666": {
+      "content": "</tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151667": {
+      "content": "<think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151668": {
+      "content": "</think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151669": {
+      "content": "<|audio_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151670": {
+      "content": "<|audio_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151671": {
+      "content": "<tts_pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151672": {
+      "content": "<tts_text_bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151673": {
+      "content": "<tts_text_eod>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151674": {
+      "content": "<tts_text_bos_single>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151675": {
+      "content": "<|audio_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>",
+    "<|audio_start|>",
+    "<|audio_end|>",
+    "<tts_pad>",
+    "<tts_text_bos>",
+    "<tts_text_bos_single>",
+    "<|audio_pad|>"
+  ],
+  "audio_bos_token": "<|audio_start|>",
+  "audio_eos_token": "<|audio_end|>",
+  "audio_token": "<|audio_pad|>",
+  "bos_token": null,
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "errors": "replace",
+  "extra_special_tokens": {
+    "audio_bos_token": "<|audio_start|>",
+    "audio_eos_token": "<|audio_end|>",
+    "audio_token": "<|audio_pad|>",
+    "image_token": "<|image_pad|>",
+    "video_token": "<|video_pad|>",
+    "vision_bos_token": "<|vision_start|>",
+    "vision_eos_token": "<|vision_end|>"
+  },
+  "fix_mistral_regex": true,
+  "image_token": "<|image_pad|>",
+  "model_max_length": 131072,
+  "pad_token": "<|endoftext|>",
+  "processor_class": "Qwen3TTSProcessor",
+  "split_special_tokens": false,
+  "tokenizer_class": "Qwen2Tokenizer",
+  "unk_token": null,
+  "video_token": "<|video_pad|>",
+  "vision_bos_token": "<|vision_start|>",
+  "vision_eos_token": "<|vision_end|>"
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff

vocence_config.yaml ADDED Viewed

	@@ -0,0 +1,16 @@

+runtime:
+  adapter: "qwen3_tts_repo_snapshot"
+  device_preference: "cuda"
+  dtype: "bfloat16"
+  default_language: "English"
+  use_flash_attention_2: false
+generation:
+  sample_rate: 24000
+  max_seconds: 30
+limits:
+  max_text_chars: 2000
+  max_instruction_chars: 600
+  default_language: "English"