Feature Extraction
Transformers
Safetensors
sentence-transformers
Chinese
English
qwen3_5
image-text-to-text
multimodal-embedding
text-embedding
image-embedding
video-embedding
mrl
custom_code
Instructions to use tencent/WeMM-Embedding-2B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use tencent/WeMM-Embedding-2B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("feature-extraction", model="tencent/WeMM-Embedding-2B", trust_remote_code=True)# Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("tencent/WeMM-Embedding-2B", trust_remote_code=True) model = AutoModelForMultimodalLM.from_pretrained("tencent/WeMM-Embedding-2B", trust_remote_code=True, device_map="auto") - sentence-transformers
How to use tencent/WeMM-Embedding-2B with sentence-transformers:
from sentence_transformers import SentenceTransformer model = SentenceTransformer("tencent/WeMM-Embedding-2B", trust_remote_code=True) sentences = [ "The weather is lovely today.", "It's so sunny outside!", "He drove to the stadium." ] embeddings = model.encode(sentences) similarities = model.similarity(embeddings, embeddings) print(similarities.shape) # [3, 3] - Notebooks
- Google Colab
- Kaggle
File size: 1,983 Bytes
0d5cf66 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 | #!/usr/bin/env python3
"""Align SGLang 0.5.9 video preprocessing with WeMM-Embedding."""
import importlib.metadata
import py_compile
import shutil
from pathlib import Path
import sglang
EXPECTED_VERSION = "0.5.9"
REPLACEMENTS = (
(
"IMAGE_FACTOR = 28",
"IMAGE_FACTOR = 32 # patch_size=16 * spatial_merge_size=2",
),
(
" idx = np.linspace(0, total_frames - 1, num=nframes, dtype=np.int64)",
" idx = torch.linspace(0, total_frames - 1, nframes).round().long().cpu().numpy()",
),
(
""" video = torchvision.transforms.functional.resize(
video,
[resized_height, resized_width],
interpolation=InterpolationMode.BILINEAR,
)
""",
""" video = torchvision.transforms.functional.resize(
video,
[resized_height, resized_width],
interpolation=InterpolationMode.BICUBIC,
antialias=True,
).float()
""",
),
)
def main() -> None:
version = importlib.metadata.version("sglang")
if version != EXPECTED_VERSION:
raise RuntimeError(f"Expected sglang=={EXPECTED_VERSION}, found {version}")
path = (
Path(sglang.__file__).resolve().parent
/ "srt"
/ "multimodal"
/ "processors"
/ "qwen_vl.py"
)
text = path.read_text(encoding="utf-8")
if all(new in text for _, new in REPLACEMENTS):
print("SGLang video preprocessing is ready.")
return
updated = text
for old, new in REPLACEMENTS:
if updated.count(old) != 1:
raise RuntimeError(f"Unexpected SGLang source: {old.splitlines()[0]}")
updated = updated.replace(old, new, 1)
backup = path.with_suffix(path.suffix + ".original")
if not backup.exists():
shutil.copy2(path, backup)
path.write_text(updated, encoding="utf-8")
py_compile.compile(str(path), doraise=True)
print("SGLang video preprocessing is ready.")
if __name__ == "__main__":
main()
|