""" Presety modelů pro HF Space s 4× NVIDIA A100 80GB (320 GB VRAM, 568 GB RAM). Poznámky k hardware: - A100 je Ampere => FP8 checkpointy běží přes Marlin kernel (weight-only, compute v FP16). Funguje dobře, ale není to nativní FP8 jako na Hopperu. - Ephemeral disk Space je jen ~50 GB => velké modely NESMÍ stahovat za běhu. Řešení: mount modelu jako read-only volume (scripts/space_ctl.py mount ...). Mountnutý model nezabírá disk a je dostupný okamžitě po startu. Preset = doporučená kombinace nastavení; po vybrání lze jednotlivá pole dál ručně upravit v tabu Nastavení. Proč tu NENÍ moonshotai/Kimi-K2.7-Code (ověřeno 2026-07): 1.06T parametrů, oficiální checkpoint je nativně INT4 a má 595 GB — na 320 GB VRAM se nevejde ani v nejagresivnějších kvantizacích (NVFP4/3.6bit ≈ 480–560 GB). Použij ho přes hybrid režim (agent_mode=hybrid, pole kimi_model — je to výchozí hodnota); volá se přes HF router API, ne lokálně. """ # Konvence mountu: space_ctl.py mount => mount_path=/repos/ MODEL_VOLUME_ROOT = "/repos" PRESETS = { "qwen3-coder-next-fp8": { "label": "Qwen3-Coder-Next (FP8, 80B-A3B) — doporučeno", "model": "Qwen/Qwen3-Coder-Next-FP8", "quantization": "auto", "dtype": "auto", "max_model_len": 262144, "tool_call_parser": "qwen3_coder", "reasoning_parser": "", "weights_gb": 80, "notes": "Nejlepší poměr výkon/rychlost pro agentní kódování (SWE-bench " "Verified ~70 %). Hybrid attention (Qwen3Next), jen 3B aktivních " "parametrů, 256K kontext s obřím prostorem na KV cache.", }, "qwen3-coder-next": { "label": "Qwen3-Coder-Next (BF16, 80B-A3B)", "model": "Qwen/Qwen3-Coder-Next", "quantization": "none", "dtype": "bfloat16", "max_model_len": 262144, "tool_call_parser": "qwen3_coder", "reasoning_parser": "", "weights_gb": 159, "notes": "Plná přesnost bez kvantizace; na Ampere může být o něco " "rychlejší než FP8-Marlin dequant.", }, "minimax-m2.7": { "label": "MiniMax-M2.7 (FP8, 229B-A10B) — nejsilnější agentic coder", "model": "MiniMaxAI/MiniMax-M2.7", "quantization": "auto", "dtype": "auto", "max_model_len": 98304, "tool_call_parser": "minimax_m2", "reasoning_parser": "minimax_m2_append_think", "weights_gb": 230, "notes": "Vlajková loď agentního kódování (interleaved thinking, " "multi-step tool use). 230 GB vah => nutný volume mount, " "kontext ~96K. Parsery ověř proti verzi vLLM.", }, "qwen35-122b-fp8": { "label": "Qwen3.5-122B-A10B (FP8) — nejnovější generace", "model": "Qwen/Qwen3.5-122B-A10B-FP8", "quantization": "auto", "dtype": "auto", "max_model_len": 131072, "tool_call_parser": "hermes", "reasoning_parser": "qwen3", "weights_gb": 127, "notes": "General+agentic model nejnovější řady (04/2026), hybridní " "thinking. Vyžaduje vLLM s podporou Qwen3_5Moe (>=0.25).", }, "qwen35-397b-int4": { "label": "Qwen3.5-397B-A17B (GPTQ-Int4) — vlajková loď", "model": "Qwen/Qwen3.5-397B-A17B-GPTQ-Int4", "quantization": "auto", "dtype": "auto", "max_model_len": 65536, "tool_call_parser": "hermes", "reasoning_parser": "qwen3", "weights_gb": 236, "notes": "Oficiální Int4 kvant největšího Qwen3.5 (gptq_marlin na " "Ampere). Nejchytřejší model, co se na 320 GB vejde; Int4 " "mírně sráží přesnost. Nutný volume mount.", }, "devstral-2-123b": { "label": "Devstral-2-123B (BF16, dense) — SWE specialista", "model": "mistralai/Devstral-2-123B-Instruct-2512", "quantization": "none", "dtype": "bfloat16", "max_model_len": 49152, "tool_call_parser": "mistral", "reasoning_parser": "", "weights_gb": 256, "notes": "Mistral agent-coder trénovaný na SWE úlohy. 256 GB vah => " "těsné (ctx ~48K), nutný mount. Ověř licenci pro tvé použití.", }, "glm-4.7-flash": { "label": "GLM-4.7-Flash (BF16, 31B) — rychlý a levný", "model": "zai-org/GLM-4.7-Flash", "quantization": "none", "dtype": "bfloat16", "max_model_len": 131072, "tool_call_parser": "glm45", "reasoning_parser": "glm45", "weights_gb": 62, "notes": "Nejnovější malý agentní model GLM řady (01/2026). " "Parser glm45 ověř proti verzi vLLM (možná glm47).", }, "qwen3-coder-30b": { "label": "Qwen3-Coder-30B-A3B (BF16) — stabilní jistota", "model": "Qwen/Qwen3-Coder-30B-A3B-Instruct", "quantization": "none", "dtype": "bfloat16", "max_model_len": 131072, "tool_call_parser": "qwen3_coder", "reasoning_parser": "", "weights_gb": 61, "notes": "MoE 30B (3B aktivních) — velmi rychlý, nativně 256K kontext, " "výborný tool-calling, dlouho prověřený ve vLLM.", }, "qwen3-coder-30b-fp8": { "label": "Qwen3-Coder-30B-A3B (FP8)", "model": "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8", "quantization": "auto", "dtype": "auto", "max_model_len": 262144, "tool_call_parser": "qwen3_coder", "reasoning_parser": "", "weights_gb": 31, "notes": "Poloviční váhy => maximální prostor pro KV cache a plný " "256K kontext. Na Ampere běží přes FP8-Marlin.", }, "qwen3-235b-fp8": { "label": "Qwen3-235B-A22B-Instruct-2507 (FP8) — vlajková loď", "model": "Qwen/Qwen3-235B-A22B-Instruct-2507-FP8", "quantization": "auto", "dtype": "auto", "max_model_len": 65536, "tool_call_parser": "hermes", "reasoning_parser": "", "weights_gb": 236, "notes": "Nejsilnější Qwen. ~236 GB vah => na 320 GB těsné, kontext " "omezen na 64K. Nutný volume mount (nevejde se na disk).", }, "glm-4.5-air-fp8": { "label": "GLM-4.5-Air (FP8)", "model": "zai-org/GLM-4.5-Air-FP8", "quantization": "auto", "dtype": "auto", "max_model_len": 131072, "tool_call_parser": "glm45", "reasoning_parser": "glm45", "weights_gb": 110, "notes": "106B MoE, silný agentní model s thinking režimem.", }, "qwen3-32b": { "label": "Qwen3-32B (dense, thinking)", "model": "Qwen/Qwen3-32B", "quantization": "none", "dtype": "bfloat16", "max_model_len": 32768, "tool_call_parser": "hermes", "reasoning_parser": "qwen3", "weights_gb": 66, "notes": "Dense reasoning model s režimem.", }, "gpt-oss-120b": { "label": "OpenAI gpt-oss-120b (MXFP4)", "model": "openai/gpt-oss-120b", "quantization": "auto", "dtype": "auto", "max_model_len": 131072, "tool_call_parser": "", "reasoning_parser": "", "weights_gb": 63, "notes": "Harmony formát (tool-calling má vLLM vestavěný, parser " "nenastavuj). Optimalizováno pro Hopper; na A100 ověř výkon.", }, } def preset_choices() -> list[tuple[str, str]]: """(label, key) páry pro Gradio dropdown.""" return [(p["label"], key) for key, p in PRESETS.items()] def preset_settings(key: str) -> dict: """Pole presetu aplikovatelná přes SettingsManager.update().""" p = PRESETS.get(key) if not p: return {} return {k: v for k, v in p.items() if k in ("model", "quantization", "dtype", "max_model_len", "tool_call_parser", "reasoning_parser")} # Fallback odvození parserů podle jména modelu, když je v nastavení "auto". _PARSER_RULES = [ ("qwen3-coder", ("qwen3_coder", "")), ("qwen3", ("hermes", "qwen3")), ("qwen2.5", ("hermes", "")), ("glm-4", ("glm45", "glm45")), ("minimax-m2", ("minimax_m2", "minimax_m2_append_think")), ("gpt-oss", ("", "")), ("kimi", ("kimi_k2", "")), ("devstral", ("mistral", "")), ("mistral", ("mistral", "")), ] def infer_parsers(model: str) -> tuple[str, str]: """Vrátí (tool_call_parser, reasoning_parser) pro daný model, jinak hermes.""" name = model.lower() for needle, parsers in _PARSER_RULES: if needle in name: return parsers return ("hermes", "")