{ "source": "OpenMOSS-Team/MOSS-Transcribe-Diarize", "format": "mlx-safetensors", "model_type": "moss_transcribe_diarize", "quantization": { "bits": 8, "group_size": 64, "mode": "affine", "scope": "text_backbone_linears_only", "excluded_prefixes": ["model.whisper_encoder", "model.vq_adaptor"], "excluded_layers": ["model.language_model.embed_tokens"], "note": "Token embedding kept full precision: mlx-audio-swift loads it as a plain Embedding and does not swap quantized embeddings, so a quantized embed_tokens fails its load-time shape check." } }