# vLLM — cau hinh cho beyoru/KAT-Coder-V2.5-Dev-VL-Flash (NVFP4). # # ⚠️ HAI CANH BAO, doc truoc khi dung: # # 1. BAT BUOC GPU BLACKWELL (sm100/sm120): B200, B300, RTX PRO 6000 Blackwell, # RTX 50-series, DGX Spark. NVFP4 chay weight 4-bit VA activation 4-bit tren # FP4 tensor core. KHONG chay tren Hopper (H100/H200), Ada, Ampere. # Tren nhung card do hay dung ban bf16: beyoru/KAT-Coder-V2.5-Dev-VL # # 2. CHUA KIEM CHUNG TREN PHAN CUNG THAT. Checkpoint duoc dung tren Hopper, noi # NVFP4 khong chay duoc. Shape weight/scale, quantization_config va cau truc # checkpoint da verify TINH; chua sinh mot token nao tren FP4 tensor core that. services: kat-vl-flash: image: vllm/vllm-openai:v0.26.0 container_name: kat-vl-flash ipc: host ports: - "8000:8000" volumes: - /path/to/models:/models - ${HOME}/.cache/huggingface:/root/.cache/huggingface deploy: resources: reservations: devices: - driver: nvidia device_ids: ["0"] capabilities: [gpu] command: > --model /models/KAT-Coder-V2.5-Dev-VL-Flash --served-model-name kat-vl-flash --max-model-len 32768 --max-num-batched-tokens 8192 --gpu-memory-utilization 0.90 --trust-remote-code # ⚠️ `--max-num-batched-tokens 8192` la BAT BUOC, dung bo di. # # Kien truc nay lai attention + Gated DeltaNet. Khi prefix caching bat (mac dinh o # vLLM moi), vLLM ep `mamba_cache_mode='align'`, roi phai nang attention block size # len 2096 de "attention page size >= mamba page size". Mac dinh # max_num_batched_tokens chi 2048 => assert vo ngay luc khoi tao KV cache: # # AssertionError: In Mamba cache align mode, # block_size (2096) must be <= max_num_batched_tokens (2048) # # Bat ky gia tri nao >= 2096 deu qua; 8192 con loi cho prefill. Cach khac la tat # prefix caching (mamba cache ve mode 'none') nhung the thi mat prefix cache — khong dang. # ⛔ DUNG ghim MoE backend. De vLLM tu chon. Tren Blackwell duong nhanh la # CUTLASS / FlashInfer-TRTLLM / CuTe-DSL; ep Marlin se tut manh vi Marlin la # kernel weight-only (W4A16), khong bao gio cham toi FP4 tensor core. # # Weight chi 22 GB (giam 68.7% so voi 70.2 GB bf16) nen vua thoai mai mot card # Blackwell don. KV cache da co scale fp8 hieu chuan san trong checkpoint. # # So do do chinh xac (doc duoc tu config.json): # NVFP4 W4A4 : toan bo mlp.experts.*.{gate,up,down}_proj + shared_expert (group 16) # FP8 W8A8 : self_attn.{q,k,v,o}_proj, linear_attn.{in_proj_qkv,in_proj_z,out_proj}, lm_head # BF16 : router mlp.gate, shared_expert_gate, linear_attn.{in_proj_a,in_proj_b,norm}, # va TOAN BO vision tower (333 tensor) # KV cache : fp8 per-tensor static