{ "validated_at": "2026-08-27", "hardware": { "gpu": "NVIDIA GB10", "compute_capability": "SM121", "memory_gib": 121.69 }, "base_runtime": { "image": "vllm/vllm-openai:glm53-flash-arm64-cu130", "digest": "sha256:905c02933be6021301db2dc284e24e3727467aa3a0f63b41d609885778a07bce", "image_id": "sha256:d42649063a2b05810bce6e1462918475b4ae6e44ad56033735f7118d5c16f136", "vllm": "0.1.dev20051+g487ecf187", "flashinfer": "0.6.17", "transformers": "5.15.1", "torch": "2.13.0+cu130" }, "schema": { "critical_config_fields_matched": 21, "mock_tensor_count": 5342, "analogous_target_tensor_names": 5342, "missing_target_tensor_names": 0, "nvfp4_weight_dtype": "U8", "nvfp4_scale_dtype": "F8_E4M3", "nvfp4_global_scale_dtype": "F32" }, "runtime": { "adapter_image_runtime_tested_local_id": "sha256:7e79066623ddc0485c71cc01a181637cc287457d86c74676043d93bc478ea491", "published_image_local_id": "sha256:0d255dd6c0f4c797214a185706fc57cf982e85952945904414eaa83e15fc1e71", "published_image": "ghcr.io/cyijun/glm-5.3-flash-nvfp4-gb10@sha256:4251b561d111d817765ed4097512ce36811deac071a4a7411d20242df5c74a47", "published_source_revision": "5bb0a598829839a9e0c420c6b737a742e084948c", "published_manifest_visibility": "public", "health": "pass", "openai_models": "pass", "short_chat_decode": "pass", "prefill_491_tokens_and_decode_16_tokens": "pass", "mtp_one_speculative_token": "pass", "note": "Generated text is intentionally random; validation concerns loading, cache layout, CUDA dispatch, prefill, decode, and MTP execution." } }