{ "candidates": { "IQ4_NL": { "bytes": 4526279776, "execution_passed": true, "mean_kld_nats": { "mean": 0.149734, "standard_error": 0.003381 }, "not_published": true, "ppl": 12.48105, "ppl_difference": 0.612923, "reason": "Only 28,606,464 bytes smaller than Q4_K_S while materially worse in KLD; its small KLD gain over the smaller IQ4_XS was not statistically compelling in this screen.", "same_top_percent": 84.522, "sha256": "3bc2c89844d293531e074e796cc05e6583d282214b6e43bbba474255705072b0" }, "MXFP4_MOE": { "blackwell_benchmark": { "artifact_comparison": "Q4_K_S", "generation_tokens_per_second": { "MXFP4_MOE": 255.307518, "Q4_K_S": 277.999794, "relative_change_percent": -8.162695 }, "prompt_tokens_per_second": { "MXFP4_MOE": 6391.33598, "Q4_K_S": 5422.155601, "relative_change_percent": 17.874448 }, "raw_report": "bench-Q4_K_S-vs-MXFP4_MOE.json", "repetitions": 5 }, "bytes": 4718248800, "execution_passed": true, "mean_kld_nats": { "mean": 0.267021, "standard_error": 0.006113 }, "not_published": true, "ppl": 12.708273, "ppl_difference": 0.840146, "reason": "Q4_K_S is 163,362,560 bytes smaller and substantially more faithful; MXFP4 improved prompt throughput on this Blackwell GPU but reduced generation throughput.", "same_top_percent": 79.424, "sha256": "dd23da2994871320d8c7ac0f29859303b3e1e6a24894056b6503c2459aeddf7b", "tensor_whitelist_audit": { "F32": 215, "MXFP4": 69, "Q8_0": 242, "passed": true, "structure_report": "structure-MXFP4_MOE.json" } } }, "purpose": "Transparent record of generated candidates omitted from the published model-weight frontier" }