Duplicate from VoiceNet/voiceclap-large

Browse files

Files changed (13) hide show

.gitattributes +36 -0
README.md +87 -0
chat_template.jinja +7 -0
config.json +144 -0
config_sentence_transformers.json +15 -0
generation_config.json +7 -0
model.safetensors +3 -0
modules.json +20 -0
preprocessor_config.json +31 -0
processor_config.json +116 -0
sentence_bert_config.json +48 -0
tokenizer.json +3 -0
tokenizer_config.json +31 -0

.gitattributes ADDED Viewed

	@@ -0,0 +1,36 @@

+*.7z filter=lfs diff=lfs merge=lfs -text
+*.arrow filter=lfs diff=lfs merge=lfs -text
+*.bin filter=lfs diff=lfs merge=lfs -text
+*.bz2 filter=lfs diff=lfs merge=lfs -text
+*.ckpt filter=lfs diff=lfs merge=lfs -text
+*.ftz filter=lfs diff=lfs merge=lfs -text
+*.gz filter=lfs diff=lfs merge=lfs -text
+*.h5 filter=lfs diff=lfs merge=lfs -text
+*.joblib filter=lfs diff=lfs merge=lfs -text
+*.lfs.* filter=lfs diff=lfs merge=lfs -text
+*.mlmodel filter=lfs diff=lfs merge=lfs -text
+*.model filter=lfs diff=lfs merge=lfs -text
+*.msgpack filter=lfs diff=lfs merge=lfs -text
+*.npy filter=lfs diff=lfs merge=lfs -text
+*.npz filter=lfs diff=lfs merge=lfs -text
+*.onnx filter=lfs diff=lfs merge=lfs -text
+*.ot filter=lfs diff=lfs merge=lfs -text
+*.parquet filter=lfs diff=lfs merge=lfs -text
+*.pb filter=lfs diff=lfs merge=lfs -text
+*.pickle filter=lfs diff=lfs merge=lfs -text
+*.pkl filter=lfs diff=lfs merge=lfs -text
+*.pt filter=lfs diff=lfs merge=lfs -text
+*.pth filter=lfs diff=lfs merge=lfs -text
+*.rar filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+saved_model/**/* filter=lfs diff=lfs merge=lfs -text
+*.tar.* filter=lfs diff=lfs merge=lfs -text
+*.tar filter=lfs diff=lfs merge=lfs -text
+*.tflite filter=lfs diff=lfs merge=lfs -text
+*.tgz filter=lfs diff=lfs merge=lfs -text
+*.wasm filter=lfs diff=lfs merge=lfs -text
+*.xz filter=lfs diff=lfs merge=lfs -text
+*.zip filter=lfs diff=lfs merge=lfs -text
+*.zst filter=lfs diff=lfs merge=lfs -text
+*tfevents* filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

README.md ADDED Viewed

	@@ -0,0 +1,87 @@

+---
+license: cc-by-4.0
+language:
+  - en
+library_name: sentence-transformers
+pipeline_tag: feature-extraction
+base_model: LCO-Embedding/LCO-Embedding-Omni-7B
+tags:
+  - audio
+  - speech
+  - emotion
+  - clap
+  - contrastive
+  - voice
+  - sentence-transformers
+---
+# VoiceCLAP-Large
+Voice-text contrastive embedding model — the larger of the two anchors
+released with [VoiceNet](https://huggingface.co/VoiceNet).
+VoiceCLAP-Large is a **single-tower** model: a rank-16 LoRA finetune of
+[LCO-Embedding-Omni-7B](https://huggingface.co/LCO-Embedding/LCO-Embedding-Omni-7B)
+(Qwen2.5-Omni-Thinker-7B backbone with a sentence-transformer
+last-token-pooling head) trained with the symmetric InfoNCE loss. The audio
+and text embeddings are produced by the same backbone — the modality is
+determined by what is fed in via the multimodal chat template.
+| | |
+| --- | --- |
+| Architecture | single-tower Omni-Embedding (Qwen2.5-Omni-Thinker-7B + ST last-token-pool) |
+| Adaptation | rank-16 LoRA (alpha 32, dropout 0.05), merged into the released weights |
+| Joint embedding | 3 584-d, L2-normalised |
+| Loss | symmetric InfoNCE (all-gather negatives) |
+| Total parameters | ~7 B (full merged model) |
+| Epochs | 1 |
+## Training data
+Trained for **1 epoch** on the open `voiceclap_10_safe` mixture (9 datasets)
+used in the VoiceNet paper:
+- `emolia-balanced-5M-subset` (annotated subset of [Emilia](https://huggingface.co/datasets/amphion/Emilia-Dataset))
+- `laions_got_talent_clean_with_captions`
+- `majestrino-data`
+- `synthetic_vocal_bursts`
+- `improved_synthetic_vocal_bursts`
+- `ears`
+- `expresso`
+- `voxceleb1`
+- `voxceleb2`
+All clips are captioned with `MOSS-Audio-8B-Thinking`-derived dense
+vocal-style captions covering emotions, talking-style attributes, and
+demographics.
+## Standalone load example
+The model uses the SentenceTransformer multimodal API — both
+`sentence-transformers` and `transformers` are on PyPI; no other deps are
+required.
+```python
+from sentence_transformers import SentenceTransformer
+model = SentenceTransformer("VoiceNet/voiceclap-large", trust_remote_code=True)
+# Text embedding (3 584-d, L2-normalised)
+text_emb = model.encode(["a calm and steady voice"])
+# Audio embedding — pass a dict with raw samples + sampling rate.
+import soundfile as sf
+arr, sr = sf.read("clip.wav")
+audio_emb = model.encode([{"array": arr, "sampling_rate": sr}])
+# Cosine similarity (embeddings already L2-normalised)
+print((audio_emb @ text_emb.T).item())
+```
+For convenience the LoRA adapter is also shipped under `adapter/` so it can
+be reapplied to other LCO-Embedding-Omni-7B forks; the merged
+`model.safetensors` already contains it.
+## Citation
+If you use this model, please cite the VoiceNet paper.

chat_template.jinja ADDED Viewed

	@@ -0,0 +1,7 @@

+{% set audio_count = namespace(value=0) %}{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system
+You are a helpful assistant.<|im_end|>
+{% endif %}<|im_start|>{{ message['role'] }}
+{% if message['content'] is string %}{{ message['content'] }}<|im_end|>
+{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_bos|><|IMAGE|><|vision_eos|>{% elif content['type'] == 'audio' or 'audio' in content or 'audio_url' in content %}{% set audio_count.value = audio_count.value + 1 %}{% if add_audio_id %}Audio {{ audio_count.value }}: {% endif %}<|audio_bos|><|AUDIO|><|audio_eos|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_bos|><|VIDEO|><|vision_eos|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>
+{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant
+{% endif %}

config.json ADDED Viewed

	@@ -0,0 +1,144 @@

+{
+  "_attn_implementation_autoset": true,
+  "architectures": [
+    "Qwen2_5OmniThinkerForConditionalGeneration"
+  ],
+  "audio_config": {
+    "_attn_implementation_autoset": true,
+    "activation_dropout": 0.0,
+    "activation_function": "gelu",
+    "attention_dropout": 0.0,
+    "d_model": 1280,
+    "dropout": 0.0,
+    "dtype": "bfloat16",
+    "encoder_attention_heads": 20,
+    "encoder_ffn_dim": 5120,
+    "encoder_layerdrop": 0.0,
+    "encoder_layers": 32,
+    "init_std": 0.02,
+    "initializer_range": 0.02,
+    "max_source_positions": 1500,
+    "model_type": "qwen2_5_omni_audio_encoder",
+    "n_window": 100,
+    "num_hidden_layers": 32,
+    "num_mel_bins": 128,
+    "output_dim": 3584,
+    "scale_embedding": false,
+    "tf_legacy_loss": false,
+    "use_bfloat16": false
+  },
+  "audio_end_token_id": 151648,
+  "audio_start_token_id": 151647,
+  "audio_token_index": 151646,
+  "bos_token_id": 151644,
+  "dtype": "bfloat16",
+  "eos_token_id": 151645,
+  "ignore_index": -100,
+  "image_token_index": 151655,
+  "init_std": 0.02,
+  "initializer_range": 0.02,
+  "model_type": "qwen2_5_omni_thinker",
+  "pad_token_id": 151643,
+  "position_id_per_seconds": 25,
+  "seconds_per_chunk": 2,
+  "text_config": {
+    "attention_dropout": 0.0,
+    "bos_token_id": null,
+    "dtype": "bfloat16",
+    "eos_token_id": null,
+    "hidden_act": "silu",
+    "hidden_size": 3584,
+    "init_std": 0.02,
+    "initializer_range": 0.02,
+    "intermediate_size": 18944,
+    "layer_types": [
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention",
+      "full_attention"
+    ],
+    "max_position_embeddings": 32768,
+    "max_window_layers": 28,
+    "model_type": "qwen2_5_omni_text",
+    "num_attention_heads": 28,
+    "num_hidden_layers": 28,
+    "num_key_value_heads": 4,
+    "pad_token_id": null,
+    "rms_norm_eps": 1e-06,
+    "rope_parameters": {
+      "mrope_section": [
+        16,
+        24,
+        24
+      ],
+      "rope_theta": 1000000.0,
+      "rope_type": "default",
+      "type": "default"
+    },
+    "sliding_window": null,
+    "use_cache": true,
+    "use_sliding_window": false,
+    "vocab_size": 152064
+  },
+  "tie_word_embeddings": false,
+  "transformers_version": "5.1.0",
+  "user_token_id": 872,
+  "video_token_index": 151656,
+  "vision_config": {
+    "_attn_implementation_autoset": true,
+    "depth": 32,
+    "dtype": "bfloat16",
+    "embed_dim": 1280,
+    "fullatt_block_indexes": [
+      7,
+      15,
+      23,
+      31
+    ],
+    "hidden_act": "silu",
+    "hidden_size": 1280,
+    "in_channels": 3,
+    "in_chans": 3,
+    "init_std": 0.02,
+    "initializer_range": 0.02,
+    "intermediate_size": 3420,
+    "model_type": "qwen2_5_omni_vision_encoder",
+    "num_heads": 16,
+    "out_hidden_size": 3584,
+    "patch_size": 14,
+    "spatial_merge_size": 2,
+    "spatial_patch_size": 14,
+    "temporal_patch_size": 2,
+    "tf_legacy_loss": false,
+    "tokens_per_second": 25,
+    "use_bfloat16": false,
+    "window_size": 112
+  },
+  "vision_end_token_id": 151653,
+  "vision_start_token_id": 151652,
+  "vision_token_id": 151654
+}

config_sentence_transformers.json ADDED Viewed

	@@ -0,0 +1,15 @@

+{
+  "__version__": {
+    "pytorch": "2.10.0+cu128",
+    "sentence_transformers": "5.4.1",
+    "transformers": "5.1.0"
+  },
+  "default_prompt_name": "default",
+  "model_type": "SentenceTransformer",
+  "prompts": {
+    "default": "You are Qwen, a virtual human developed by the Qwen Team, Alibaba Group, capable of perceiving auditory and visual inputs, as well as generating text and speech.",
+    "document": "",
+    "query": ""
+  },
+  "similarity_fn_name": "cosine"
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,7 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 151644,
+  "eos_token_id": 151645,
+  "pad_token_id": 151643,
+  "transformers_version": "5.1.0"
+}

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:5cb1c7c101cd5c360775bde52742bc7369bb8906d1991c4ab3cbeab836d81bf0
+size 18133945952

modules.json ADDED Viewed

	@@ -0,0 +1,20 @@

+[
+  {
+    "idx": 0,
+    "name": "0",
+    "path": "",
+    "type": "sentence_transformers.base.modules.transformer.Transformer"
+  },
+  {
+    "idx": 1,
+    "name": "1",
+    "path": "1_Pooling",
+    "type": "sentence_transformers.sentence_transformer.modules.pooling.Pooling"
+  },
+  {
+    "idx": 2,
+    "name": "2",
+    "path": "2_Normalize",
+    "type": "sentence_transformers.sentence_transformer.modules.normalize.Normalize"
+  }
+]

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "chunk_length": 300,
+  "dither": 0.0,
+  "feature_extractor_type": "WhisperFeatureExtractor",
+  "feature_size": 128,
+  "hop_length": 160,
+  "image_mean": [
+    0.48145466,
+    0.4578275,
+    0.40821073
+  ],
+  "image_processor_type": "Qwen2VLImageProcessor",
+  "image_std": [
+    0.26862954,
+    0.26130258,
+    0.27577711
+  ],
+  "max_pixels": 12845056,
+  "merge_size": 2,
+  "min_pixels": 3136,
+  "n_fft": 400,
+  "n_samples": 4800000,
+  "nb_max_frames": 30000,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "patch_size": 14,
+  "processor_class": "Qwen2_5OmniProcessor",
+  "return_attention_mask": true,
+  "sampling_rate": 16000,
+  "temporal_patch_size": 2
+}

processor_config.json ADDED Viewed

	@@ -0,0 +1,116 @@

+{
+  "feature_extractor": {
+    "chunk_length": 300,
+    "dither": 0.0,
+    "feature_extractor_type": "WhisperFeatureExtractor",
+    "feature_size": 128,
+    "hop_length": 160,
+    "image_mean": [
+      0.48145466,
+      0.4578275,
+      0.40821073
+    ],
+    "image_processor_type": "Qwen2VLImageProcessor",
+    "image_std": [
+      0.26862954,
+      0.26130258,
+      0.27577711
+    ],
+    "max_pixels": 12845056,
+    "merge_size": 2,
+    "min_pixels": 3136,
+    "n_fft": 400,
+    "n_samples": 4800000,
+    "nb_max_frames": 30000,
+    "padding_side": "right",
+    "padding_value": 0.0,
+    "patch_size": 14,
+    "return_attention_mask": true,
+    "sampling_rate": 16000,
+    "temporal_patch_size": 2
+  },
+  "image_processor": {
+    "chunk_length": 300,
+    "data_format": "channels_first",
+    "dither": 0.0,
+    "do_convert_rgb": true,
+    "do_normalize": true,
+    "do_rescale": true,
+    "do_resize": true,
+    "feature_size": 128,
+    "hop_length": 160,
+    "image_mean": [
+      0.48145466,
+      0.4578275,
+      0.40821073
+    ],
+    "image_processor_type": "Qwen2VLImageProcessorFast",
+    "image_std": [
+      0.26862954,
+      0.26130258,
+      0.27577711
+    ],
+    "merge_size": 2,
+    "n_fft": 400,
+    "n_samples": 4800000,
+    "nb_max_frames": 30000,
+    "padding_side": "right",
+    "padding_value": 0.0,
+    "patch_size": 14,
+    "resample": 3,
+    "rescale_factor": 0.00392156862745098,
+    "return_attention_mask": true,
+    "sampling_rate": 16000,
+    "size": {
+      "longest_edge": 12845056,
+      "shortest_edge": 3136
+    },
+    "temporal_patch_size": 2
+  },
+  "processor_class": "Qwen2_5OmniProcessor",
+  "video_processor": {
+    "chunk_length": 300,
+    "data_format": "channels_first",
+    "default_to_square": true,
+    "dither": 0.0,
+    "do_convert_rgb": true,
+    "do_normalize": true,
+    "do_rescale": true,
+    "do_resize": true,
+    "do_sample_frames": false,
+    "feature_extractor_type": "WhisperFeatureExtractor",
+    "feature_size": 128,
+    "hop_length": 160,
+    "image_mean": [
+      0.48145466,
+      0.4578275,
+      0.40821073
+    ],
+    "image_processor_type": "Qwen2VLImageProcessor",
+    "image_std": [
+      0.26862954,
+      0.26130258,
+      0.27577711
+    ],
+    "max_frames": 768,
+    "merge_size": 2,
+    "min_frames": 4,
+    "n_fft": 400,
+    "n_samples": 4800000,
+    "nb_max_frames": 30000,
+    "padding_side": "right",
+    "padding_value": 0.0,
+    "patch_size": 14,
+    "resample": 3,
+    "rescale_factor": 0.00392156862745098,
+    "return_attention_mask": true,
+    "return_metadata": false,
+    "sampling_rate": 16000,
+    "size": {
+      "longest_edge": 12845056,
+      "shortest_edge": 3136
+    },
+    "temporal_patch_size": 2,
+    "video_processor_type": "Qwen2VLVideoProcessor"
+  }
+}

sentence_bert_config.json ADDED Viewed

	@@ -0,0 +1,48 @@

+{
+    "transformer_task": "any-to-any",
+    "modality_config": {
+        "text": {
+            "method": "forward",
+            "method_output_name": [
+                "hidden_states",
+                -1
+            ]
+        },
+        "image": {
+            "method": "forward",
+            "method_output_name": [
+                "hidden_states",
+                -1
+            ]
+        },
+        "audio": {
+            "method": "forward",
+            "method_output_name": [
+                "hidden_states",
+                -1
+            ]
+        },
+        "video": {
+            "method": "forward",
+            "method_output_name": [
+                "hidden_states",
+                -1
+            ]
+        },
+        "message": {
+            "method": "forward",
+            "method_output_name": [
+                "hidden_states",
+                -1
+            ],
+            "format": "structured"
+        }
+    },
+    "module_output_name": "token_embeddings",
+    "processing_kwargs": {
+        "chat_template": {
+            "chat_template": "sentence_transformers",
+            "add_generation_prompt": true
+        }
+    }
+}

tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0c75e6d39795d574e5ec741e767ca690ea08a33bfa024ef2d372b4e4c72db191
+size 11422137

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "add_prefix_space": false,
+  "audio_bos_token": "<|audio_bos|>",
+  "audio_eos_token": "<|audio_eos|>",
+  "audio_token": "<|AUDIO|>",
+  "backend": "tokenizers",
+  "bos_token": null,
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "errors": "replace",
+  "image_token": "<|IMAGE|>",
+  "is_local": true,
+  "model_max_length": 32768,
+  "model_specific_special_tokens": {
+    "audio_bos_token": "<|audio_bos|>",
+    "audio_eos_token": "<|audio_eos|>",
+    "audio_token": "<|AUDIO|>",
+    "image_token": "<|IMAGE|>",
+    "video_token": "<|VIDEO|>",
+    "vision_bos_token": "<|vision_bos|>",
+    "vision_eos_token": "<|vision_eos|>"
+  },
+  "pad_token": "<|endoftext|>",
+  "processor_class": "Qwen2_5OmniProcessor",
+  "split_special_tokens": false,
+  "tokenizer_class": "TokenizersBackend",
+  "unk_token": null,
+  "video_token": "<|VIDEO|>",
+  "vision_bos_token": "<|vision_bos|>",
+  "vision_eos_token": "<|vision_eos|>"
+}