"""Testy VLLMEngine — sestavení příkazu, resolve modelu, životní cyklus.""" import time def _settings(settings_module, **overrides): s = settings_module.Settings() for k, v in overrides.items(): setattr(s, k, v) return s def test_build_command_defaults(engine_module, settings_module): eng = engine_module.VLLMEngine() s = _settings(settings_module) # default Qwen3-Coder-Next-FP8 cmd = eng.build_command(s) assert cmd[1] == "serve" assert cmd[2] == "Qwen/Qwen3-Coder-Next-FP8" assert "--tensor-parallel-size" in cmd and cmd[cmd.index("--tensor-parallel-size") + 1] == "4" # auto parser => qwen3_coder, bez reasoning parseru assert "--enable-auto-tool-choice" in cmd assert cmd[cmd.index("--tool-call-parser") + 1] == "qwen3_coder" assert "--reasoning-parser" not in cmd # quantization auto => žádný --quantization flag assert "--quantization" not in cmd assert "--enforce-eager" not in cmd assert "--no-enable-prefix-caching" not in cmd def test_build_command_parser_inference(engine_module, settings_module): eng = engine_module.VLLMEngine() cmd = eng.build_command(_settings(settings_module, model="Qwen/Qwen3-32B")) assert cmd[cmd.index("--tool-call-parser") + 1] == "hermes" assert cmd[cmd.index("--reasoning-parser") + 1] == "qwen3" cmd = eng.build_command(_settings(settings_module, model="zai-org/GLM-4.5-Air-FP8")) assert cmd[cmd.index("--tool-call-parser") + 1] == "glm45" # gpt-oss: harmony má vLLM vestavěné => žádný tool parser cmd = eng.build_command(_settings(settings_module, model="openai/gpt-oss-120b")) assert "--tool-call-parser" not in cmd assert "--enable-auto-tool-choice" not in cmd def test_build_command_explicit_flags(engine_module, settings_module): eng = engine_module.VLLMEngine() s = _settings(settings_module, quantization="fp8", kv_cache_dtype="fp8", enforce_eager=True, enable_prefix_caching=False, max_num_seqs=16, tool_call_parser="hermes", reasoning_parser="", engine_extra_args="--swap-space 8") cmd = eng.build_command(s) assert cmd[cmd.index("--quantization") + 1] == "fp8" assert cmd[cmd.index("--kv-cache-dtype") + 1] == "fp8" assert "--enforce-eager" in cmd assert "--no-enable-prefix-caching" in cmd assert cmd[cmd.index("--max-num-seqs") + 1] == "16" assert cmd[cmd.index("--tool-call-parser") + 1] == "hermes" assert "--reasoning-parser" not in cmd assert "--swap-space" in cmd and cmd[cmd.index("--swap-space") + 1] == "8" def test_download_dir_explicit_setting(engine_module, settings_module, tmp_path): eng = engine_module.VLLMEngine() target = tmp_path / "bucket" / "models" s = _settings(settings_module, model="org/hub-model", download_dir=str(target)) cmd = eng.build_command(s) assert cmd[cmd.index("--download-dir") + 1] == str(target) assert eng.active_download_dir == str(target) assert target.is_dir() # adresář se vytvořil def test_download_dir_bucket_autodetect(engine_module, settings_module, tmp_path, monkeypatch): # bez explicitního nastavení: zapisovatelný "/data" (bucket) má přednost data = tmp_path / "data" data.mkdir() monkeypatch.setattr(engine_module, "BUCKET_DATA_DIR", str(data)) s = _settings(settings_module, model="org/hub-model", download_dir="") resolved = engine_module.resolve_download_dir(s) assert resolved == data / "models" # bez bucketu spadne na ephemeral default monkeypatch.setattr(engine_module, "BUCKET_DATA_DIR", str(tmp_path / "neni")) resolved = engine_module.resolve_download_dir(s) assert str(resolved) == engine_module.DEFAULT_DOWNLOAD_DIR def test_download_dir_not_used_for_mounted_model(engine_module, settings_module, tmp_path, monkeypatch): # model z RO volume mountu se nestahuje => žádný --download-dir volume_root = tmp_path / "repos" (volume_root / "org" / "mounted").mkdir(parents=True) monkeypatch.setattr(engine_module, "MODEL_VOLUME_ROOT", str(volume_root)) eng = engine_module.VLLMEngine() cmd = eng.build_command(_settings(settings_module, model="org/mounted", download_dir=str(tmp_path / "dl"))) assert "--download-dir" not in cmd assert eng.active_download_dir is None def test_tp_clamped_to_available_gpus(engine_module, settings_module, monkeypatch): """TP=4 na 1 GPU nesmí shodit vLLM (ValidationError) — snižuje se.""" monkeypatch.setattr(engine_module, "available_gpu_count", lambda: 1) eng = engine_module.VLLMEngine() cmd = eng.build_command(_settings(settings_module, tensor_parallel_size=4)) assert cmd[cmd.index("--tensor-parallel-size") + 1] == "1" st_keys = eng.status_dict() assert st_keys["detected_gpus"] == 1 assert st_keys["effective_tensor_parallel"] == 1 assert st_keys["tensor_parallel_clamped_from"] == 4 def test_tp_unchanged_when_gpus_sufficient(engine_module, settings_module, monkeypatch): monkeypatch.setattr(engine_module, "available_gpu_count", lambda: 4) eng = engine_module.VLLMEngine() cmd = eng.build_command(_settings(settings_module, tensor_parallel_size=4)) assert cmd[cmd.index("--tensor-parallel-size") + 1] == "4" assert eng.status_dict()["tensor_parallel_clamped_from"] is None def test_tp_unchanged_when_gpu_count_unknown(engine_module, settings_module, monkeypatch): # lokální vývoj bez GPU: 0 = nezjištěno => TP se nemění monkeypatch.setattr(engine_module, "available_gpu_count", lambda: 0) eng = engine_module.VLLMEngine() cmd = eng.build_command(_settings(settings_module, tensor_parallel_size=4)) assert cmd[cmd.index("--tensor-parallel-size") + 1] == "4" def test_available_gpu_count_from_cvd(engine_module, monkeypatch): monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0,1,2,3") assert engine_module.available_gpu_count() == 4 monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0") assert engine_module.available_gpu_count() == 1 monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "") assert engine_module.available_gpu_count() == 0 def test_build_command_revision_pinning(engine_module, settings_module): eng = engine_module.VLLMEngine() cmd = eng.build_command(_settings(settings_module, model="org/hub-model", model_revision="abc123def")) assert cmd[cmd.index("--revision") + 1] == "abc123def" # bez pinu žádný --revision cmd = eng.build_command(_settings(settings_module, model="org/hub-model", model_revision="")) assert "--revision" not in cmd # lokální cesta => revision se nepoužije cmd = eng.build_command(_settings(settings_module, model="/mnt/model", model_revision="abc123def")) assert "--revision" not in cmd def test_build_child_env(engine_module, tmp_path, monkeypatch): base = { "PATH": "/usr/bin", "VLLM_BUILD_COMMIT": "deadbeef", "VLLM_BUILD_PIPELINE": "x", "VLLM_IMAGE_TAG": "v0.25.1", "HF_HUB_ENABLE_HF_TRANSFER": "1", "HF_HOME": "/app/cache/hf", } # bez bucketu: šum pryč, Xet zapnut, HF_HOME nedotčen monkeypatch.setattr(engine_module, "BUCKET_DATA_DIR", str(tmp_path / "neni")) env = engine_module.build_child_env(base) assert "VLLM_BUILD_COMMIT" not in env and "VLLM_IMAGE_TAG" not in env assert "HF_HUB_ENABLE_HF_TRANSFER" not in env assert env["HF_XET_HIGH_PERFORMANCE"] == "1" assert env["VLLM_WORKER_MULTIPROC_METHOD"] == "spawn" assert env["HF_HOME"] == "/app/cache/hf" assert "VLLM_CACHE_ROOT" not in env # s RW bucketem: persistentní cache pro torch.compile artefakty i HF data = tmp_path / "data" data.mkdir() monkeypatch.setattr(engine_module, "BUCKET_DATA_DIR", str(data)) env = engine_module.build_child_env({}) assert env["VLLM_CACHE_ROOT"] == str(data / "cache" / "vllm") assert env["HF_HOME"] == str(data / "cache" / "hf") # explicitní hodnoty mají přednost (setdefault) env = engine_module.build_child_env({"VLLM_CACHE_ROOT": "/muj"}) assert env["VLLM_CACHE_ROOT"] == "/muj" def test_engine_api_key_not_fixed(engine_module): assert engine_module.ENGINE_API_KEY != "codeagent-internal" assert len(engine_module.ENGINE_API_KEY) >= 16 def test_resolve_model(engine_module, tmp_path, monkeypatch): # 1) absolutní cesta path, source = engine_module.resolve_model("/nekde/model") assert (path, source) == ("/nekde/model", "path") # 2) volume mount /repos/ volume_root = tmp_path / "repos" (volume_root / "org" / "model-a").mkdir(parents=True) monkeypatch.setattr(engine_module, "MODEL_VOLUME_ROOT", str(volume_root)) path, source = engine_module.resolve_model("org/model-a") assert source == "volume" and path.endswith("org/model-a") # 3) lokální adresář /app/models local_root = tmp_path / "models" (local_root / "org--model-b").mkdir(parents=True) monkeypatch.setattr(engine_module, "LOCAL_MODEL_DIR", local_root) path, source = engine_module.resolve_model("org/model-b") assert source == "local" and path.endswith("org--model-b") # 4) fallback: HF hub path, source = engine_module.resolve_model("org/unknown") assert (path, source) == ("org/unknown", "hub") def test_lifecycle_start_ready_stop(engine_module, settings_module): eng = engine_module.VLLMEngine() s = _settings(settings_module, model="org/lifecycle-model") eng.remember_settings(s) eng.start(s, block=True) try: assert eng.is_ready, eng.status_dict() st = eng.status_dict() assert st["model"] == "org/lifecycle-model" assert st["pid"] is not None # OpenAI klient proti mock serveru funguje client = eng.openai_client() resp = client.chat.completions.create( model="code-agent-llm", messages=[{"role": "user", "content": "ping"}]) assert resp.choices[0].message.content == "mock:org/lifecycle-model:ping" finally: eng.stop() assert eng.status_dict()["state"] == "stopped" assert eng.status_dict()["pid"] is None def test_reload_swaps_model(engine_module, settings_module): eng = engine_module.VLLMEngine() eng.start(_settings(settings_module, model="org/model-one"), block=True) try: assert eng.is_ready eng.reload(_settings(settings_module, model="org/model-two"), block=True) assert eng.is_ready, eng.status_dict() assert eng.current_model == "org/model-two" client = eng.openai_client() resp = client.chat.completions.create( model="code-agent-llm", messages=[{"role": "user", "content": "x"}]) assert "org/model-two" in resp.choices[0].message.content finally: eng.stop() def test_start_failure_sets_error(engine_module, settings_module, monkeypatch): monkeypatch.setenv("MOCK_VLLM_FAIL", "1") eng = engine_module.VLLMEngine() eng.start(_settings(settings_module, model="org/crash"), block=True) st = eng.status_dict() assert st["state"] == "error" assert "exit" in (st["last_error"] or "") eng.stop()