liuqingquan commited on Jan 7

Commit

e798605

verified ·

1 Parent(s): ee3a9ef

Upload folder using huggingface_hub

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

.gitattributes +1 -0
models/TTS/IndexTTS-1.5/.gitattributes +35 -0
models/TTS/IndexTTS-1.5/README +5 -0
models/TTS/IndexTTS-1.5/README.md +3 -0
models/TTS/IndexTTS-1.5/bigvgan_discriminator.pth +3 -0
models/TTS/IndexTTS-1.5/bigvgan_generator.pth +3 -0
models/TTS/IndexTTS-1.5/config.yaml +113 -0
models/TTS/IndexTTS-1.5/dvae.pth +3 -0
models/TTS/IndexTTS-1.5/gpt.pth +3 -0
models/TTS/IndexTTS-1.5/unigram_12000.vocab +0 -0
models/TTS/IndexTTS-2/.msc +0 -0
models/TTS/IndexTTS-2/.mv +1 -0
models/TTS/IndexTTS-2/README.md +74 -0
models/TTS/IndexTTS-2/bpe.model +3 -0
models/TTS/IndexTTS-2/config.yaml +120 -0
models/TTS/IndexTTS-2/configuration.json +1 -0
models/TTS/IndexTTS-2/feat1.pt +3 -0
models/TTS/IndexTTS-2/feat2.pt +3 -0
models/TTS/IndexTTS-2/gpt.pth +3 -0
models/TTS/IndexTTS-2/qwen0.6bemo4-merge/Modelfile +11 -0
models/TTS/IndexTTS-2/qwen0.6bemo4-merge/added_tokens.json +28 -0
models/TTS/IndexTTS-2/qwen0.6bemo4-merge/chat_template.jinja +4 -0
models/TTS/IndexTTS-2/qwen0.6bemo4-merge/config.json +30 -0
models/TTS/IndexTTS-2/qwen0.6bemo4-merge/generation_config.json +6 -0
models/TTS/IndexTTS-2/qwen0.6bemo4-merge/merges.txt +0 -0
models/TTS/IndexTTS-2/qwen0.6bemo4-merge/model.safetensors +3 -0
models/TTS/IndexTTS-2/qwen0.6bemo4-merge/special_tokens_map.json +31 -0
models/TTS/IndexTTS-2/qwen0.6bemo4-merge/tokenizer.json +3 -0
models/TTS/IndexTTS-2/qwen0.6bemo4-merge/tokenizer_config.json +240 -0
models/TTS/IndexTTS-2/qwen0.6bemo4-merge/vocab.json +0 -0
models/TTS/IndexTTS-2/s2mel.pth +3 -0
models/TTS/IndexTTS-2/wav2vec2bert_stats.pt +3 -0
models/TTS/IndexTTS/bigvgan_discriminator.pth +3 -0
models/TTS/IndexTTS/bigvgan_generator.pth +3 -0
models/TTS/IndexTTS/dvae.pth +3 -0
models/TTS/IndexTTS/gpt.pth +3 -0
models/TTS/IndexTTS/unigram_12000.vocab +0 -0
models/TTS/MaskGCT/semantic_codec/model.safetensors +3 -0
models/TTS/bigvgan_v2_22khz_80band_256x/bigvgan_generator.pt +3 -0
models/TTS/bigvgan_v2_22khz_80band_256x/config.json +63 -0
models/TTS/campplus/campplus_cn_common.bin +3 -0
models/TTS/openaudio-s1-mini/.msc +0 -0
models/TTS/openaudio-s1-mini/.mv +1 -0
models/TTS/openaudio-s1-mini/README.md +92 -0
models/TTS/openaudio-s1-mini/codec.pth +3 -0
models/TTS/openaudio-s1-mini/config.json +32 -0
models/TTS/openaudio-s1-mini/configuration.json +1 -0
models/TTS/openaudio-s1-mini/model.pth +3 -0
models/TTS/openaudio-s1-mini/special_tokens.json +0 -0
models/TTS/openaudio-s1-mini/tokenizer.tiktoken +0 -0

.gitattributes CHANGED Viewed

@@ -46,3 +46,4 @@ models/unet/FLUX/flux1-dev.sft filter=lfs diff=lfs merge=lfs -text
 models/unet/IC-Light/iclight_sd15_fcon.safetensors.crdownload filter=lfs diff=lfs merge=lfs -text
 models/unet/wan2.1/Wan2.2_T2V_High_Noise_14B_VACE-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
 models/unet/wan2.1/Wan2.2_T2V_Low_Noise_14B_VACE-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text

 models/unet/IC-Light/iclight_sd15_fcon.safetensors.crdownload filter=lfs diff=lfs merge=lfs -text
 models/unet/wan2.1/Wan2.2_T2V_High_Noise_14B_VACE-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
 models/unet/wan2.1/Wan2.2_T2V_Low_Noise_14B_VACE-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
+models/TTS/IndexTTS-2/qwen0.6bemo4-merge/tokenizer.json filter=lfs diff=lfs merge=lfs -text

models/TTS/IndexTTS-1.5/.gitattributes ADDED Viewed

	@@ -0,0 +1,35 @@

+*.7z filter=lfs diff=lfs merge=lfs -text
+*.arrow filter=lfs diff=lfs merge=lfs -text
+*.bin filter=lfs diff=lfs merge=lfs -text
+*.bz2 filter=lfs diff=lfs merge=lfs -text
+*.ckpt filter=lfs diff=lfs merge=lfs -text
+*.ftz filter=lfs diff=lfs merge=lfs -text
+*.gz filter=lfs diff=lfs merge=lfs -text
+*.h5 filter=lfs diff=lfs merge=lfs -text
+*.joblib filter=lfs diff=lfs merge=lfs -text
+*.lfs.* filter=lfs diff=lfs merge=lfs -text
+*.mlmodel filter=lfs diff=lfs merge=lfs -text
+*.model filter=lfs diff=lfs merge=lfs -text
+*.msgpack filter=lfs diff=lfs merge=lfs -text
+*.npy filter=lfs diff=lfs merge=lfs -text
+*.npz filter=lfs diff=lfs merge=lfs -text
+*.onnx filter=lfs diff=lfs merge=lfs -text
+*.ot filter=lfs diff=lfs merge=lfs -text
+*.parquet filter=lfs diff=lfs merge=lfs -text
+*.pb filter=lfs diff=lfs merge=lfs -text
+*.pickle filter=lfs diff=lfs merge=lfs -text
+*.pkl filter=lfs diff=lfs merge=lfs -text
+*.pt filter=lfs diff=lfs merge=lfs -text
+*.pth filter=lfs diff=lfs merge=lfs -text
+*.rar filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+saved_model/**/* filter=lfs diff=lfs merge=lfs -text
+*.tar.* filter=lfs diff=lfs merge=lfs -text
+*.tar filter=lfs diff=lfs merge=lfs -text
+*.tflite filter=lfs diff=lfs merge=lfs -text
+*.tgz filter=lfs diff=lfs merge=lfs -text
+*.wasm filter=lfs diff=lfs merge=lfs -text
+*.xz filter=lfs diff=lfs merge=lfs -text
+*.zip filter=lfs diff=lfs merge=lfs -text
+*.zst filter=lfs diff=lfs merge=lfs -text
+*tfevents* filter=lfs diff=lfs merge=lfs -text

models/TTS/IndexTTS-1.5/README ADDED Viewed

	@@ -0,0 +1,5 @@

+大更新(效果很不错）：
+1. 大幅增加了英文训练数据，提升英文及跨语种合成效果；
+2. 增大模型参数至0.5B左右；
+3. wer, ss 及 韵律都有明显的提升；
+4. gpt输出：text token 和 mel token 是连在一起的。

models/TTS/IndexTTS-1.5/README.md ADDED Viewed

	@@ -0,0 +1,3 @@

+---
+license: apache-2.0
+---

models/TTS/IndexTTS-1.5/bigvgan_discriminator.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:46e1f6277f7239363d2393f2f9fe36902cf8995e4acc0ba67ed25a025dbd02f0
+size 1651507545

models/TTS/IndexTTS-1.5/bigvgan_generator.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a2458834d8277e76eb8614c9751b5e8eaa0474eab706f0ecfafcb600023133ed
+size 536176992

models/TTS/IndexTTS-1.5/config.yaml ADDED Viewed

	@@ -0,0 +1,113 @@

+dataset:
+    bpe_model: bpe.model
+    sample_rate: 24000
+    squeeze: false
+    mel:
+        sample_rate: 24000
+        n_fft: 1024
+        hop_length: 256
+        win_length: 1024
+        n_mels: 100
+        mel_fmin: 0
+        normalize: false
+gpt:
+    model_dim: 1280
+    max_mel_tokens: 800
+    max_text_tokens: 600
+    heads: 20
+    use_mel_codes_as_input: true
+    mel_length_compression: 1024
+    layers: 24
+    number_text_tokens: 12000
+    number_mel_codes: 8194
+    start_mel_token: 8192
+    stop_mel_token: 8193
+    start_text_token: 0
+    stop_text_token: 1
+    train_solo_embeddings: false
+    condition_type: "conformer_perceiver"
+    condition_module:
+        output_size: 512
+        linear_units: 2048
+        attention_heads: 8
+        num_blocks: 6
+        input_layer: "conv2d2"
+        perceiver_mult: 2
+vqvae:
+    channels: 100
+    num_tokens: 8192
+    hidden_dim: 512
+    num_resnet_blocks: 3
+    codebook_dim: 512
+    num_layers: 2
+    positional_dims: 1
+    kernel_size: 3
+    smooth_l1_loss: true
+    use_transposed_convs: false
+bigvgan:
+    adam_b1: 0.8
+    adam_b2: 0.99
+    lr_decay: 0.999998
+    seed: 1234
+    resblock: "1"
+    upsample_rates: [4,4,4,4,2,2]
+    upsample_kernel_sizes: [8,8,4,4,4,4]
+    upsample_initial_channel: 1536
+    resblock_kernel_sizes: [3,7,11]
+    resblock_dilation_sizes: [[1,3,5], [1,3,5], [1,3,5]]
+    feat_upsample: false
+    speaker_embedding_dim: 512
+    cond_d_vector_in_each_upsampling_layer: true
+    gpt_dim: 1280
+    activation: "snakebeta"
+    snake_logscale: true
+    use_cqtd_instead_of_mrd: true
+    cqtd_filters: 128
+    cqtd_max_filters: 1024
+    cqtd_filters_scale: 1
+    cqtd_dilations: [1, 2, 4]
+    cqtd_hop_lengths: [512, 256, 256]
+    cqtd_n_octaves: [9, 9, 9]
+    cqtd_bins_per_octaves: [24, 36, 48]
+    resolutions: [[1024, 120, 600], [2048, 240, 1200], [512, 50, 240]]
+    mpd_reshapes: [2, 3, 5, 7, 11]
+    use_spectral_norm: false
+    discriminator_channel_mult: 1
+    use_multiscale_melloss: true
+    lambda_melloss: 15
+    clip_grad_norm: 1000
+    segment_size: 16384
+    num_mels: 100
+    num_freq: 1025
+    n_fft: 1024
+    hop_size: 256
+    win_size: 1024
+    sampling_rate: 24000
+    fmin: 0
+    fmax: null
+    fmax_for_loss: null
+    mel_type: "pytorch"
+    num_workers: 2
+    dist_config:
+        dist_backend: "nccl"
+        dist_url: "tcp://localhost:54321"
+        world_size: 1
+dvae_checkpoint: dvae.pth
+gpt_checkpoint: gpt.pth
+bigvgan_checkpoint: bigvgan_generator.pth
+version: 1.5

models/TTS/IndexTTS-1.5/dvae.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:69e841bf8cd97a32806ea8a439c50017c991ac9e8bb795db89ec47828cae4d5d
+size 243316270

models/TTS/IndexTTS-1.5/gpt.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:44460b820a8afd58f68f3d3e69113e7900c8730bf519ecf158c081f2b8991240
+size 1171228980

models/TTS/IndexTTS-1.5/unigram_12000.vocab ADDED Viewed

The diff for this file is too large to render. See raw diff

models/TTS/IndexTTS-2/.msc ADDED Viewed

Binary file (1.86 kB). View file

models/TTS/IndexTTS-2/.mv ADDED Viewed

	@@ -0,0 +1 @@


1	+ Revision:master,CreatedAt:1757355497

models/TTS/IndexTTS-2/README.md ADDED Viewed

	@@ -0,0 +1,74 @@

+---
+license: apache-2.0
+language:
+- en
+---
+## 👉🏻 IndexTTS2 👈🏻
+<center><h3>IndexTTS2: A Breakthrough in Emotionally Expressive and Duration-Controlled Auto-Regressive Zero-Shot Text-to-Speech</h3></center>
+<div align="center">
+  <a href='https://arxiv.org/abs/2506.21619'>
+    <img src='https://img.shields.io/badge/ArXiv-2506.21619-red?logo=arxiv'/>
+  </a>
+  <br/>
+  <a href='https://github.com/index-tts/index-tts'>
+    <img src='https://img.shields.io/badge/GitHub-Code-orange?logo=github'/>
+  </a>
+  <a href='https://index-tts.github.io/index-tts2.github.io/'>
+    <img src='https://img.shields.io/badge/GitHub-Demo-orange?logo=github'/>
+  </a>
+  <br/>
+  <!--a href='https://huggingface.co/spaces/IndexTeam/IndexTTS'>
+    <img src='https://img.shields.io/badge/HuggingFace-Demo-blue?logo=huggingface'/>
+  </a-->
+  <a href='https://huggingface.co/IndexTeam/IndexTTS-2'>
+    <img src='https://img.shields.io/badge/HuggingFace-Model-blue?logo=huggingface' />
+  </a>
+  <br/>
+  <!--a href='https://modelscope.cn/studios/IndexTeam/IndexTTS-Demo'>
+    <img src='https://img.shields.io/badge/ModelScope-Demo-purple?logo=modelscope'/>
+  </a-->
+  <a href='https://modelscope.cn/models/IndexTeam/IndexTTS-2'>
+    <img src='https://img.shields.io/badge/ModelScope-Model-purple?logo=modelscope'/>
+  </a>
+</div>
+## Acknowledge
+1. [tortoise-tts](https://github.com/neonbjb/tortoise-tts)
+2. [XTTSv2](https://github.com/coqui-ai/TTS)
+3. [BigVGAN](https://github.com/NVIDIA/BigVGAN)
+4. [wenet](https://github.com/wenet-e2e/wenet/tree/main)
+5. [icefall](https://github.com/k2-fsa/icefall)
+6. [maskgct](https://github.com/open-mmlab/Amphion/tree/main/models/tts/maskgct)
+7. [seed-vc](https://github.com/Plachtaa/seed-vc)
+## 📚 Citation
+🌟 If you find our work helpful, please leave us a star and cite our paper.
+IndexTTS2
+```
+@article{zhou2025indextts2,
+  title={IndexTTS2: A Breakthrough in Emotionally Expressive and Duration-Controlled Auto-Regressive Zero-Shot Text-to-Speech},
+  author={Siyi Zhou, Yiquan Zhou, Yi He, Xun Zhou, Jinchao Wang, Wei Deng, Jingchen Shu},
+  journal={arXiv preprint arXiv:2506.21619},
+  year={2025}
+}
+```
+IndexTTS
+```
+@article{deng2025indextts,
+  title={IndexTTS: An Industrial-Level Controllable and Efficient Zero-Shot Text-To-Speech System},
+  author={Wei Deng, Siyi Zhou, Jingchen Shu, Jinchao Wang, Lu Wang},
+  journal={arXiv preprint arXiv:2502.05512},
+  year={2025},
+  doi={10.48550/arXiv.2502.05512},
+  url={https://arxiv.org/abs/2502.05512}
+}
+```

models/TTS/IndexTTS-2/bpe.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b2a5ce8090d32da3642cc4f81fdc996376bc6dd3f4cd5e3d165f71120d9f2bc8
+size 475997

models/TTS/IndexTTS-2/config.yaml ADDED Viewed

	@@ -0,0 +1,120 @@

+dataset:
+    bpe_model: bpe.model
+    sample_rate: 24000
+    squeeze: false
+    mel:
+        sample_rate: 24000
+        n_fft: 1024
+        hop_length: 256
+        win_length: 1024
+        n_mels: 100
+        mel_fmin: 0
+        normalize: false
+gpt:
+    model_dim: 1280
+    max_mel_tokens: 1815
+    max_text_tokens: 600
+    heads: 20
+    use_mel_codes_as_input: true
+    mel_length_compression: 1024
+    layers: 24
+    number_text_tokens: 12000
+    number_mel_codes: 8194
+    start_mel_token: 8192
+    stop_mel_token: 8193
+    start_text_token: 0
+    stop_text_token: 1
+    train_solo_embeddings: false
+    condition_type: "conformer_perceiver"
+    condition_module:
+        output_size: 512
+        linear_units: 2048
+        attention_heads: 8
+        num_blocks: 6
+        input_layer: "conv2d2"
+        perceiver_mult: 2
+    emo_condition_module:
+        output_size: 512
+        linear_units: 1024
+        attention_heads: 4
+        num_blocks: 4
+        input_layer: "conv2d2"
+        perceiver_mult: 2
+semantic_codec:
+    codebook_size: 8192
+    hidden_size: 1024
+    codebook_dim: 8
+    vocos_dim: 384
+    vocos_intermediate_dim: 2048
+    vocos_num_layers: 12
+s2mel:
+    preprocess_params:
+        sr: 22050
+        spect_params:
+            n_fft: 1024
+            win_length: 1024
+            hop_length: 256
+            n_mels: 80
+            fmin: 0
+            fmax: "None"
+    dit_type: "DiT"
+    reg_loss_type: "l1"
+    style_encoder:
+        dim: 192
+    length_regulator:
+        channels: 512
+        is_discrete: false
+        in_channels: 1024
+        content_codebook_size: 2048
+        sampling_ratios: [1, 1, 1, 1]
+        vector_quantize: false
+        n_codebooks: 1
+        quantizer_dropout: 0.0
+        f0_condition: false
+        n_f0_bins: 512
+    DiT:
+        hidden_dim: 512
+        num_heads: 8
+        depth: 13
+        class_dropout_prob: 0.1
+        block_size: 8192
+        in_channels: 80
+        style_condition: true
+        final_layer_type: 'wavenet'
+        target: 'mel'
+        content_dim: 512
+        content_codebook_size: 1024
+        content_type: 'discrete'
+        f0_condition: false
+        n_f0_bins: 512
+        content_codebooks: 1
+        is_causal: false
+        long_skip_connection: true
+        zero_prompt_speech_token: false
+        time_as_token: false
+        style_as_token: false
+        uvit_skip_connection: true
+        add_resblock_in_transformer: false
+    wavenet:
+        hidden_dim: 512
+        num_layers: 8
+        kernel_size: 5
+        dilation_rate: 1
+        p_dropout: 0.2
+        style_condition: true
+gpt_checkpoint: gpt.pth
+w2v_stat: wav2vec2bert_stats.pt
+s2mel_checkpoint: s2mel.pth
+emo_matrix: feat2.pt
+spk_matrix: feat1.pt
+emo_num: [3, 17, 2, 8, 4, 5, 10, 24]
+qwen_emo_path: qwen0.6bemo4-merge/
+vocoder:
+    type: "bigvgan"
+    name: "nvidia/bigvgan_v2_22khz_80band_256x"
+version: 2.0

models/TTS/IndexTTS-2/configuration.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"task":"text-to-speech"}

models/TTS/IndexTTS-2/feat1.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f219cb447d80216ba615666da2ff8d63ac544eee26657f3a7b278692bf7a67c4
+size 57170

models/TTS/IndexTTS-2/feat2.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9c4292e96dee535aea9a6206e9a0c856dd578dde9212acdb16dd3ada4d12bf80
+size 374866

models/TTS/IndexTTS-2/gpt.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:baaaeb8b56328da81731dc540a85a7dee32eca9da28f174b05757cb651c602a4
+size 3484663079

models/TTS/IndexTTS-2/qwen0.6bemo4-merge/Modelfile ADDED Viewed

	@@ -0,0 +1,11 @@

+# ollama modelfile auto-generated by llamafactory
+FROM .
+TEMPLATE """{{ if .System }}System: {{ .System }}<|endoftext|>
+{{ end }}{{ range .Messages }}{{ if eq .Role "user" }}Human: {{ .Content }}<|endoftext|>
+Assistant:{{ else if eq .Role "assistant" }}{{ .Content }}<|endoftext|>
+{{ end }}{{ end }}"""
+PARAMETER stop "<|endoftext|>"
+PARAMETER num_ctx 4096

models/TTS/IndexTTS-2/qwen0.6bemo4-merge/added_tokens.json ADDED Viewed

	@@ -0,0 +1,28 @@

+{
+  "</think>": 151668,
+  "</tool_call>": 151658,
+  "</tool_response>": 151666,
+  "<think>": 151667,
+  "<tool_call>": 151657,
+  "<tool_response>": 151665,
+  "<|box_end|>": 151649,
+  "<|box_start|>": 151648,
+  "<|endoftext|>": 151643,
+  "<|file_sep|>": 151664,
+  "<|fim_middle|>": 151660,
+  "<|fim_pad|>": 151662,
+  "<|fim_prefix|>": 151659,
+  "<|fim_suffix|>": 151661,
+  "<|im_end|>": 151645,
+  "<|im_start|>": 151644,
+  "<|image_pad|>": 151655,
+  "<|object_ref_end|>": 151647,
+  "<|object_ref_start|>": 151646,
+  "<|quad_end|>": 151651,
+  "<|quad_start|>": 151650,
+  "<|repo_name|>": 151663,
+  "<|video_pad|>": 151656,
+  "<|vision_end|>": 151653,
+  "<|vision_pad|>": 151654,
+  "<|vision_start|>": 151652
+}

models/TTS/IndexTTS-2/qwen0.6bemo4-merge/chat_template.jinja ADDED Viewed

	@@ -0,0 +1,4 @@

+{% if messages[0]['role'] == 'system' %}{% set loop_messages = messages[1:] %}{% set system_message = messages[0]['content'] %}{% else %}{% set loop_messages = messages %}{% endif %}{% if system_message is defined %}{{ 'System: ' + system_message + '<|endoftext|>' + '
+' }}{% endif %}{% for message in loop_messages %}{% set content = message['content'] %}{% if message['role'] == 'user' %}{{ 'Human: ' + content + '<|endoftext|>' + '
+Assistant:' }}{% elif message['role'] == 'assistant' %}{{ content + '<|endoftext|>' + '
+' }}{% endif %}{% endfor %}

models/TTS/IndexTTS-2/qwen0.6bemo4-merge/config.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "architectures": [
+    "Qwen3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "bos_token_id": 151643,
+  "eos_token_id": 151643,
+  "head_dim": 128,
+  "hidden_act": "silu",
+  "hidden_size": 1024,
+  "initializer_range": 0.02,
+  "intermediate_size": 3072,
+  "max_position_embeddings": 32768,
+  "max_window_layers": 28,
+  "model_type": "qwen3",
+  "num_attention_heads": 16,
+  "num_hidden_layers": 28,
+  "num_key_value_heads": 8,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 1000000,
+  "sliding_window": null,
+  "tie_word_embeddings": true,
+  "torch_dtype": "bfloat16",
+  "transformers_version": "4.52.1",
+  "use_cache": true,
+  "use_sliding_window": false,
+  "vocab_size": 151936
+}

models/TTS/IndexTTS-2/qwen0.6bemo4-merge/generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "bos_token_id": 151643,
+  "eos_token_id": 151643,
+  "max_new_tokens": 2048,
+  "transformers_version": "4.52.1"
+}

models/TTS/IndexTTS-2/qwen0.6bemo4-merge/merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

models/TTS/IndexTTS-2/qwen0.6bemo4-merge/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:11293257a8df593c154a8ecd5fc039f3076de35411e35f06d41b471e136f6641
+size 1192135096

models/TTS/IndexTTS-2/qwen0.6bemo4-merge/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>"
+  ],
+  "eos_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

models/TTS/IndexTTS-2/qwen0.6bemo4-merge/tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:aeb13307a71acd8fe81861d94ad54ab689df773318809eed3cbe794b4492dae4
+size 11422654

models/TTS/IndexTTS-2/qwen0.6bemo4-merge/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,240 @@

+{
+  "add_bos_token": false,
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "151643": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151644": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151645": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151646": {
+      "content": "<|object_ref_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151647": {
+      "content": "<|object_ref_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151648": {
+      "content": "<|box_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151649": {
+      "content": "<|box_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151650": {
+      "content": "<|quad_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151651": {
+      "content": "<|quad_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151652": {
+      "content": "<|vision_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151653": {
+      "content": "<|vision_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151654": {
+      "content": "<|vision_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151655": {
+      "content": "<|image_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151656": {
+      "content": "<|video_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151657": {
+      "content": "<tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151658": {
+      "content": "</tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151659": {
+      "content": "<|fim_prefix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151660": {
+      "content": "<|fim_middle|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151661": {
+      "content": "<|fim_suffix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151662": {
+      "content": "<|fim_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151663": {
+      "content": "<|repo_name|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151664": {
+      "content": "<|file_sep|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151665": {
+      "content": "<tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151666": {
+      "content": "</tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151667": {
+      "content": "<think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151668": {
+      "content": "</think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>"
+  ],
+  "bos_token": null,
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|endoftext|>",
+  "errors": "replace",
+  "extra_special_tokens": {},
+  "model_max_length": 131072,
+  "pad_token": "<|endoftext|>",
+  "padding_side": "left",
+  "split_special_tokens": false,
+  "tokenizer_class": "Qwen2Tokenizer",
+  "unk_token": null
+}

models/TTS/IndexTTS-2/qwen0.6bemo4-merge/vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff

models/TTS/IndexTTS-2/s2mel.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:aae1bb12017cbb47e7a5ce537fc82f40b6b1deb71acdb9b8f25686f32714b636
+size 1202198223

models/TTS/IndexTTS-2/wav2vec2bert_stats.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c9c176c2b8850ab2e3ba828bbfa969deaf4566ce55db5f2687b8430b87526ad2
+size 9343

models/TTS/IndexTTS/bigvgan_discriminator.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8a11c977d56c2500c7978affd08678da7a217af124356d88010fa2abcbf51984
+size 1629487449

models/TTS/IndexTTS/bigvgan_generator.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9ec77084929fad053355669c8b5986e32542f13afeff78ad93389a8f06ce62b0
+size 525166944

models/TTS/IndexTTS/dvae.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c112404dfe25d8d88084b507b0637037a419b4a5a0d9160516d9398a8f2b52c8
+size 243316270

models/TTS/IndexTTS/gpt.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:7797ed691d9c0295fd30af153d9ff04501e353a4c67c3f898e4b0840a5ef10dd
+size 696529044

models/TTS/IndexTTS/unigram_12000.vocab ADDED Viewed

The diff for this file is too large to render. See raw diff

models/TTS/MaskGCT/semantic_codec/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ec947271175d8cad75ec37e83aa487e27c97a0f72a303393772da5ffa84bddf2
+size 177183712

models/TTS/bigvgan_v2_22khz_80band_256x/bigvgan_generator.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e95ba25972d3de0628d99cd156e9315a9c018899bf739988959ebe3544080ced
+size 449228171

models/TTS/bigvgan_v2_22khz_80band_256x/config.json ADDED Viewed

	@@ -0,0 +1,63 @@

+{
+    "resblock": "1",
+    "num_gpus": 0,
+    "batch_size": 32,
+    "learning_rate": 0.0001,
+    "adam_b1": 0.8,
+    "adam_b2": 0.99,
+    "lr_decay": 0.9999996,
+    "seed": 1234,
+    "upsample_rates": [4,4,2,2,2,2],
+    "upsample_kernel_sizes": [8,8,4,4,4,4],
+    "upsample_initial_channel": 1536,
+    "resblock_kernel_sizes": [3,7,11],
+    "resblock_dilation_sizes": [[1,3,5], [1,3,5], [1,3,5]],
+    "use_tanh_at_final": false,
+    "use_bias_at_final": false,
+    "activation": "snakebeta",
+    "snake_logscale": true,
+    "use_cqtd_instead_of_mrd": true,
+    "cqtd_filters": 128,
+    "cqtd_max_filters": 1024,
+    "cqtd_filters_scale": 1,
+    "cqtd_dilations": [1, 2, 4],
+    "cqtd_hop_lengths": [512, 256, 256],
+    "cqtd_n_octaves": [9, 9, 9],
+    "cqtd_bins_per_octaves": [24, 36, 48],
+    "mpd_reshapes": [2, 3, 5, 7, 11],
+    "use_spectral_norm": false,
+    "discriminator_channel_mult": 1,
+    "use_multiscale_melloss": true,
+    "lambda_melloss": 15,
+    "clip_grad_norm": 500,
+    "segment_size": 65536,
+    "num_mels": 80,
+    "num_freq": 1025,
+    "n_fft": 1024,
+    "hop_size": 256,
+    "win_size": 1024,
+    "sampling_rate": 22050,
+    "fmin": 0,
+    "fmax": null,
+    "fmax_for_loss": null,
+    "normalize_volume": true,
+    "num_workers": 4,
+    "dist_config": {
+        "dist_backend": "nccl",
+        "dist_url": "tcp://localhost:54321",
+        "world_size": 1
+    }
+}

models/TTS/campplus/campplus_cn_common.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:3388cf5fd3493c9ac9c69851d8e7a8badcfb4f3dc631020c4961371646d5ada8
+size 28036335

models/TTS/openaudio-s1-mini/.msc ADDED Viewed

Binary file (585 Bytes). View file

models/TTS/openaudio-s1-mini/.mv ADDED Viewed

	@@ -0,0 +1 @@


1	+ Revision:master,CreatedAt:1749083191

models/TTS/openaudio-s1-mini/README.md ADDED Viewed

	@@ -0,0 +1,92 @@

+---
+tags:
+- text-to-speech
+license: cc-by-nc-sa-4.0
+language:
+- zh
+- en
+- de
+- ja
+- fr
+- es
+- ko
+- ar
+- nl
+- ru
+- it
+- pl
+- pt
+pipeline_tag: text-to-speech
+inference: false
+extra_gated_prompt: >-
+  You agree to not use the model to generate contents that violate DMCA or local
+  laws.
+extra_gated_fields:
+  Country: country
+  Specific date: date_picker
+  I agree to use this model for non-commercial use ONLY: checkbox
+---
+# OpenAudio S1
+**OpenAudio S1** is a leading text-to-speech (TTS) model trained on more than 2 million hours of audio data in multiple languages.
+Supported languages:
+- English (en)
+- Chinese (zh)
+- Japanese (ja)
+- German (de)
+- French (fr)
+- Spanish (es)
+- Korean (ko)
+- Arabic (ar)
+- Russian (ru)
+- Dutch (nl)
+- Italian (it)
+- Polish (pl)
+- Portuguese (pt)
+Please refer to [Fish Speech Github](https://github.com/fishaudio/fish-speech) for more info.
+Demo available at [Fish Audio Playground](https://fish.audio).
+Visit the [OpenAudio website](https://openaudio.com) for blog & tech report.
+## Emotion and Tone Support
+OpenAudio S1 supports a variety of emotional, tone, and special markers to enhance speech synthesis:
+**1. Emotional markers:**
+(angry) (sad) (disdainful) (excited) (surprised) (satisfied) (unhappy) (anxious) (hysterical) (delighted) (scared) (worried) (indifferent) (upset) (impatient) (nervous) (guilty) (scornful) (frustrated) (depressed) (panicked) (furious) (empathetic) (embarrassed) (reluctant) (disgusted) (keen) (moved) (proud) (relaxed) (grateful) (confident) (interested) (curious) (confused) (joyful) (disapproving) (negative) (denying) (astonished) (serious) (sarcastic) (conciliative) (comforting) (sincere) (sneering) (hesitating) (yielding) (painful) (awkward) (amused)
+**2. Tone markers:**
+(in a hurry tone) (shouting) (screaming) (whispering) (soft tone)
+**3. Special markers:**
+(laughing) (chuckling) (sobbing) (crying loudly) (sighing) (panting) (groaning) (crowd laughing) (background laughter) (audience laughing)
+**Special markers with corresponding onomatopoeia:**
+- Laughing: Ha,ha,ha
+- Chuckling: Hmm,hmm
+## Model Variants and Performance
+OpenAudio S1 includes the following models:
+-   **S1 (4B, proprietary):** The full-sized model.
+-   **S1-mini (0.5B):** A distilled version of S1.
+Both S1 and S1-mini incorporate online Reinforcement Learning from Human Feedback (RLHF).
+**Seed TTS Eval Metrics (English, auto eval, based on OpenAI gpt-4o-transcribe, speaker distance using Revai/pyannote-wespeaker-voxceleb-resnet34-LM):**
+-   **S1:**
+    -   WER (Word Error Rate): **0.008**
+    -   CER (Character Error Rate): **0.004**
+    -   Distance: **0.332**
+-   **S1-mini:**
+    -   WER (Word Error Rate): **0.011**
+    -   CER (Character Error Rate): **0.005**
+    -   Distance: **0.380**
+## License
+This model is permissively licensed under the CC-BY-NC-SA-4.0 license.

models/TTS/openaudio-s1-mini/codec.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:74fc41c5a7151c6f350af8bd7e5d6e3accfcc7f3dfbfac23afd35af07052bb2f
+size 1871099728

models/TTS/openaudio-s1-mini/config.json ADDED Viewed

	@@ -0,0 +1,32 @@

+{
+    "attention_o_bias": false,
+    "attention_qk_norm": true,
+    "attention_qkv_bias": false,
+    "codebook_size": 4096,
+    "dim": 1024,
+    "dropout": 0.0,
+    "fast_attention_o_bias": false,
+    "fast_attention_qk_norm": false,
+    "fast_attention_qkv_bias": false,
+    "fast_dim": 1024,
+    "fast_head_dim": 64,
+    "fast_intermediate_size": 3072,
+    "fast_n_head": 16,
+    "fast_n_local_heads": 8,
+    "head_dim": 128,
+    "initializer_range": 0.03125,
+    "intermediate_size": 3072,
+    "max_seq_len": 8192,
+    "model_type": "dual_ar",
+    "n_fast_layer": 4,
+    "n_head": 16,
+    "n_layer": 28,
+    "n_local_heads": 8,
+    "norm_eps": 1e-06,
+    "num_codebooks": 10,
+    "rope_base": 1000000,
+    "scale_codebook_embeddings": true,
+    "tie_word_embeddings": false,
+    "use_gradient_checkpointing": true,
+    "vocab_size": 155776
+}

models/TTS/openaudio-s1-mini/configuration.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"framework": "pytorch", "task": "text-to-speech", "allow_remote": true}

models/TTS/openaudio-s1-mini/model.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9e59be7dc6714040dce3cde1f41e730c2f0daa5339785b1cd3b60041208c35e6
+size 1735122974

models/TTS/openaudio-s1-mini/special_tokens.json ADDED Viewed

The diff for this file is too large to render. See raw diff

models/TTS/openaudio-s1-mini/tokenizer.tiktoken ADDED Viewed

The diff for this file is too large to render. See raw diff