Download tests/test_runtime_accelerator.py from User1342/distinct: direct link, hf CLI and curl.
- Browser
- Download file 10.3 kB
-
https://huggingface.co/spaces/User1342/distinct/resolve/main/tests/test_runtime_accelerator.py
- Command line
-
hf download hf://spaces/User1342/distinct/tests/test_runtime_accelerator.py
-
curl -L -o test_runtime_accelerator.py https://huggingface.co/spaces/User1342/distinct/resolve/main/tests/test_runtime_accelerator.py
10.3 kB
| """Picking the build that suits the machine, and stepping down when it does not. | |
| The benchmark that prompted all of this ran twenty agent workloads on a laptop | |
| with a CUDA card, at 3.1 tokens per second, with the GPU sitting at its 7 W idle | |
| draw. Nothing was broken: the only recorded build was the CPU one, so the CPU | |
| did the work. These are the properties that stop that happening silently, and | |
| the ones that stop the cure being worse -- a CUDA build installed on a machine | |
| that cannot start it is a worker that does not run at all, which is worse than a | |
| slow one. | |
| """ | |
| from __future__ import annotations | |
| import io | |
| import json | |
| import zipfile | |
| from pathlib import Path | |
| import pytest | |
| from distinct_agent import runtime | |
| CUDA_URL = "https://github.com/ggml-org/llama.cpp/releases/download/bT/cuda.zip" | |
| CUDART_URL = "https://github.com/ggml-org/llama.cpp/releases/download/bT/cudart.zip" | |
| CPU_URL = "https://github.com/ggml-org/llama.cpp/releases/download/bT/cpu.zip" | |
| def _zip(*names: str) -> bytes: | |
| buf = io.BytesIO() | |
| with zipfile.ZipFile(buf, "w") as handle: | |
| for name in names: | |
| handle.writestr(f"build/bin/{name}", b"x" * 16) | |
| return buf.getvalue() | |
| def _sha(payload: bytes) -> str: | |
| import hashlib | |
| return hashlib.sha256(payload).hexdigest() | |
| def _serve(bodies: dict[str, bytes]): | |
| """An opener that answers each pinned URL with its own archive.""" | |
| class _Response: | |
| def __init__(self, payload: bytes) -> None: | |
| self.headers = {"Content-Length": str(len(payload))} | |
| self._buffer = io.BytesIO(payload) | |
| def read(self, size=-1): | |
| return self._buffer.read(size) | |
| def __enter__(self): | |
| return self | |
| def __exit__(self, *exc): | |
| return False | |
| def opener(request, timeout=None): | |
| url = request.full_url if hasattr(request, "full_url") else str(request) | |
| if url not in bodies: | |
| raise AssertionError(f"unpinned URL was fetched: {url}") | |
| return _Response(bodies[url]) | |
| return opener | |
| def _pins(tmp_path: Path, entries: dict) -> Path: | |
| path = tmp_path / "pins.json" | |
| path.write_text(json.dumps({"builds": entries}), encoding="utf-8") | |
| return path | |
| def pinned(tmp_path: Path): | |
| cuda, cudart, cpu = _zip("llama-server.exe"), _zip("cudart64_12.dll"), _zip("llama-server.exe") | |
| path = _pins( | |
| tmp_path, | |
| { | |
| "test-x64": { | |
| "tag": "bT", | |
| "asset": "cpu.zip", | |
| "url": CPU_URL, | |
| "sha256": _sha(cpu), | |
| "bytes": len(cpu), | |
| }, | |
| "test-x64-cuda": { | |
| "tag": "bT", | |
| "accelerator": "cuda", | |
| "requires_cuda": 12040, | |
| "archives": [ | |
| {"asset": "cuda.zip", "url": CUDA_URL, "sha256": _sha(cuda), "bytes": len(cuda)}, | |
| { | |
| "asset": "cudart.zip", | |
| "url": CUDART_URL, | |
| "sha256": _sha(cudart), | |
| "bytes": len(cudart), | |
| }, | |
| ], | |
| }, | |
| }, | |
| ) | |
| return path, _serve({CUDA_URL: cuda, CUDART_URL: cudart, CPU_URL: cpu}) | |
| def test_a_cuda_build_is_offered_only_when_the_driver_can_run_it(pinned, monkeypatch) -> None: | |
| path, _ = pinned | |
| monkeypatch.setattr(runtime, "cuda_capability", lambda: 12040) | |
| assert runtime.preferred_keys("test-x64", path) == ["test-x64-cuda", "test-x64"] | |
| def test_a_driver_below_the_pinned_cuda_version_gets_the_cpu_build(pinned, monkeypatch) -> None: | |
| path, _ = pinned | |
| monkeypatch.setattr(runtime, "cuda_capability", lambda: 12000) | |
| assert runtime.preferred_keys("test-x64", path) == ["test-x64"] | |
| def test_a_machine_with_no_nvidia_card_is_never_offered_cuda(pinned, monkeypatch) -> None: | |
| path, _ = pinned | |
| monkeypatch.setattr(runtime, "cuda_capability", lambda: 0) | |
| assert runtime.preferred_keys("test-x64", path) == ["test-x64"] | |
| def test_the_probe_is_not_run_when_no_accelerated_build_is_recorded(tmp_path, monkeypatch) -> None: | |
| """No NVML call, no nvidia-smi subprocess, on the machines that cannot use one.""" | |
| asked = [] | |
| monkeypatch.setattr(runtime, "cuda_capability", lambda: asked.append(1) or 0) | |
| path = _pins(tmp_path, {"test-x64": {"tag": "bT", "url": CPU_URL, "sha256": "a" * 64}}) | |
| assert runtime.preferred_keys("test-x64", path) == ["test-x64"] | |
| assert asked == [] | |
| def test_an_accelerated_pin_installs_every_archive_it_names(pinned, tmp_path, monkeypatch) -> None: | |
| """The CUDA runtime is not optional: without it the executable does not load.""" | |
| path, opener = pinned | |
| monkeypatch.setattr(runtime, "cuda_capability", lambda: 12040) | |
| root = tmp_path / "proj" | |
| root.mkdir() | |
| found = runtime.ensure(root=root, key="test-x64-cuda", opener=opener, pins_path=path) | |
| directory = root / "runtime" / "test-x64-cuda" | |
| assert Path(found).parent == directory | |
| assert {item.name for item in directory.iterdir()} == {"llama-server.exe", "cudart64_12.dll"} | |
| def test_a_second_archive_that_fails_leaves_no_half_install(tmp_path, monkeypatch) -> None: | |
| cuda, cudart = _zip("llama-server.exe"), _zip("cudart64_12.dll") | |
| path = _pins( | |
| tmp_path, | |
| { | |
| "test-x64-cuda": { | |
| "tag": "bT", | |
| "accelerator": "cuda", | |
| "requires_cuda": 12040, | |
| "archives": [ | |
| {"asset": "cuda.zip", "url": CUDA_URL, "sha256": _sha(cuda)}, | |
| {"asset": "cudart.zip", "url": CUDART_URL, "sha256": "0" * 64}, | |
| ], | |
| } | |
| }, | |
| ) | |
| root = tmp_path / "proj" | |
| root.mkdir() | |
| with pytest.raises(runtime.RuntimeVerificationError): | |
| runtime.ensure( | |
| root=root, | |
| key="test-x64-cuda", | |
| opener=_serve({CUDA_URL: cuda, CUDART_URL: cudart}), | |
| pins_path=path, | |
| ) | |
| assert not (root / "runtime" / "test-x64-cuda").exists() | |
| def test_a_build_that_will_not_start_steps_down_to_the_next_one(pinned, tmp_path, monkeypatch) -> None: | |
| """A driver may advertise a version it cannot honour. Starting it is the proof.""" | |
| path, opener = pinned | |
| monkeypatch.setattr(runtime, "cuda_capability", lambda: 12040) | |
| root = tmp_path / "proj" | |
| root.mkdir() | |
| refused = [] | |
| def verify(executable: str) -> bool: | |
| if "cuda" in executable: | |
| refused.append(executable) | |
| return False | |
| return True | |
| found = runtime.ensure_best( | |
| root=root, key="test-x64", opener=opener, pins_path=path, verify=verify | |
| ) | |
| assert refused, "the accelerated build should have been tried first" | |
| assert Path(found).parent.name == "test-x64" | |
| def test_when_nothing_starts_the_reason_names_every_build_that_was_tried( | |
| pinned, tmp_path, monkeypatch | |
| ) -> None: | |
| path, opener = pinned | |
| monkeypatch.setattr(runtime, "cuda_capability", lambda: 12040) | |
| root = tmp_path / "proj" | |
| root.mkdir() | |
| with pytest.raises(runtime.RuntimeUnavailable) as caught: | |
| runtime.ensure_best( | |
| root=root, key="test-x64", opener=opener, pins_path=path, verify=lambda _: False | |
| ) | |
| assert "test-x64-cuda" in str(caught.value) | |
| assert "test-x64" in str(caught.value) | |
| def test_an_older_flat_install_is_moved_rather_than_re_downloaded(tmp_path, monkeypatch) -> None: | |
| """Builds used to unpack straight into runtime/. That install still counts.""" | |
| monkeypatch.setattr(runtime, "platform_key", lambda *a, **k: "test-x64") | |
| root = tmp_path / "proj" | |
| flat = root / "runtime" | |
| flat.mkdir(parents=True) | |
| (flat / "llama-server.exe").write_bytes(b"old") | |
| (flat / "ggml.dll").write_bytes(b"old") | |
| runtime._migrate_flat_install(root) | |
| assert not (flat / "llama-server.exe").exists() | |
| moved = flat / "test-x64" | |
| assert (moved / "llama-server.exe").read_bytes() == b"old" | |
| assert (moved / "ggml.dll").exists() | |
| assert runtime.installed_server(root, "test-x64") == str(moved / "llama-server.exe") | |
| def test_the_cpu_build_is_always_the_last_resort(pinned, monkeypatch) -> None: | |
| """Whatever else is recorded, the list ends with the build that runs anywhere.""" | |
| path, _ = pinned | |
| for capability in (0, 12000, 12040, 13000): | |
| monkeypatch.setattr(runtime, "cuda_capability", lambda c=capability: c) | |
| assert runtime.preferred_keys("test-x64", path)[-1] == "test-x64" | |
| def test_an_accelerator_this_selector_cannot_check_is_not_installed(tmp_path, monkeypatch) -> None: | |
| """A HIP pin on an unexamined machine is exactly the failure being prevented.""" | |
| payload = _zip("llama-server.exe") | |
| path = _pins( | |
| tmp_path, | |
| { | |
| "test-x64": {"tag": "bT", "url": CPU_URL, "sha256": _sha(payload)}, | |
| "test-x64-hip": {"tag": "bT", "accelerator": "hip", "url": CUDA_URL, "sha256": _sha(payload)}, | |
| }, | |
| ) | |
| assert runtime.preferred_keys("test-x64", path) == ["test-x64"] | |
| def test_a_pin_pointing_off_github_is_refused_before_anything_is_fetched(tmp_path) -> None: | |
| payload = _zip("llama-server.exe") | |
| path = _pins( | |
| tmp_path, | |
| { | |
| "test-x64-cuda": { | |
| "tag": "bT", | |
| "accelerator": "cuda", | |
| "archives": [ | |
| {"asset": "cuda.zip", "url": CUDA_URL, "sha256": _sha(payload)}, | |
| {"asset": "x.zip", "url": "https://example.invalid/x.zip", "sha256": _sha(payload)}, | |
| ], | |
| } | |
| }, | |
| ) | |
| root = tmp_path / "proj" | |
| root.mkdir() | |
| def refuse(request, timeout=None): | |
| raise AssertionError("nothing should have been fetched") | |
| with pytest.raises(runtime.RuntimeUnavailable, match="not GitHub"): | |
| runtime.ensure(root=root, key="test-x64-cuda", opener=refuse, pins_path=path) | |
| def test_the_cuda_probe_survives_a_machine_with_no_bindings_and_no_smi(monkeypatch) -> None: | |
| """Probing is best-effort by construction: it must never be the thing that fails.""" | |
| monkeypatch.setattr(runtime, "_cuda_from_nvml", lambda: 0) | |
| monkeypatch.setattr(runtime.shutil, "which", lambda name: None) | |
| assert runtime.cuda_capability() == 0 | |