{ "engine": "omnivoice", "model_type": "omnivoice", "upstream": "k2-fsa/OmniVoice", "architecture": "masked-diffusion (discrete-diffusion) codec LM; Qwen3-0.6B backbone, bidirectional attention", "audio": {"sample_rate": 24000}, "frame_rate": 25, "hop_length": 960, "num_audio_codebook": 8, "audio_vocab_size": 1025, "audio_mask_id": 1024, "audio_codebook_weights": [8, 8, 6, 6, 4, 4, 2, 2], "codec": "Higgs Audio V2 tokenizer (DAC acoustic + HuBERT semantic, 8 RVQ codebooks)", "graphs": { "model": "omnivoice_backbone.onnx", "acoustic_encoder_path": "acoustic_encoder.onnx", "semantic_encoder_path": "semantic_encoder.onnx", "quantizer_encoder_path": "quantizer_encoder.onnx", "decoder_path": "higgs_decoder.onnx", "bpe_tokenizer_path": "tokenizer.json" }, "generation_defaults": { "num_step": 32, "guidance_scale": 2.0, "t_shift": 0.1, "layer_penalty_factor": 5.0, "position_temperature": 5.0, "class_temperature": 0.0 } }