{ "ax_engine": { "decode_kernel": null, "fused_mtp": null, "kernel_evidence": "unmeasured", "model_manifest": "model-manifest.json", "preferred_group_size": 32 }, "compatible_runtimes": [ { "compatibility_level": "B", "manifest": "config.json", "mtp_support": "runtime-dependent", "name": "mlx-lm", "notes": [ "Standard backbone inference is the compatibility target.", "AXQuant MTP metadata may be ignored by MLX-LM." ], "standard_inference": true, "standard_mlx_weights": true, "support_level": "standard-inference" } ], "created_at": "2026-08-02T18:28:40.264399Z", "kv_cache": null, "memory_policy": { "kv_cache_precision": "runtime-default", "mtp_buffers": "preallocate-when-enabled", "prefix_cache": "runtime-managed", "unified_memory_safety_margin": "benchmark-required" }, "mtp": { "acceptance_retention": null, "detected": true, "draft_tokens": 1, "enabled_by_default": true, "head_precision": null, "measured_speedup": null, "optimized": false, "recommended_temperature_max": null, "sidecar_file": "mtp.safetensors", "verification_mode": "runtime-default" }, "optimization_scope": "text-path", "primary_runtime": { "compatibility_level": "A", "manifest": "model-manifest.json", "mtp_support": "native", "name": "ax-engine", "notes": [ "Runtime claims require a passing AX Engine doctor and benchmark report." ], "standard_inference": true, "standard_mlx_weights": true, "support_level": "optimized" }, "schema_version": "axquant.runtime.v1" }