| """ |
| ui/gpu.py |
| --------- |
| ZeroGPU wiring — the only place in NOVA that touches a GPU. |
| |
| HF's ZeroGPU hardware hands a Space a GPU *only* for the duration of a call to an |
| @spaces.GPU-decorated function, and refuses to boot at all if it can't find one |
| at import time ("No @spaces.GPU function detected during startup"). That single |
| constraint drives the whole design here. |
| |
| NOVA is CPU-first and stays that way. Exactly one operation runs on the GPU: the |
| bulk embedding of a paper's chunks during vectorizing, which is by far the |
| slowest thing in the app (tens of seconds on CPU, a few on GPU). Everything else |
| — SPECTER reranking, query embedding, the cross-encoder — is pinned to CPU *on |
| purpose*, because those run outside any GPU window and a cuda-resident model |
| there would fail on first use. That's why the three backend call sites now take |
| an explicit `device` instead of auto-detecting. |
| |
| Off ZeroGPU (local, or Spaces "CPU basic") @spaces.GPU is a transparent |
| passthrough and ON_ZEROGPU is False, so this module quietly degrades to plain |
| CPU work and nothing else in the app changes. |
| |
| THE CUDA-VIRGINITY RULE |
| ----------------------- |
| ZeroGPU forks its GPU worker from this process (`multiprocessing.get_context |
| ('fork')`), and the very first thing the child does is: |
| |
| os.environ['CUDA_VISIBLE_DEVICES'] = nvidia_uuid |
| torch.Tensor([0]).cuda() |
| |
| CUDA_VISIBLE_DEVICES is honoured by the driver *exactly once per process*, at |
| first CUDA init. So if anything has really initialised CUDA in THIS process |
| before the fork, the child inherits a driver that already decided there are zero |
| devices, the env var is ignored, and the worker dies with |
| |
| RuntimeError: No CUDA GPUs are available |
| |
| `spaces` prevents that by monkey-patching torch at `import spaces`. But note how |
| it does it (spaces/zero/torch/patching.py::patch): the `torch.cuda.*` attribute |
| fakes are module-global, while the TorchFunctionMode/TorchDispatchMode that |
| intercept real tensor ops are **thread-local to the thread that called patch()** |
| — the main thread, at import. The library's own source carries the TODO |
| admitting the inconsistency. Consequence for us: heavy model loading must happen |
| on the MAIN THREAD, not on Gradio worker threads or threads we spawn ourselves. |
| See ui/agents.preload_all(). |
| |
| `cuda_state()` below is the telemetry that proves whether that rule is holding. |
| """ |
| import os |
| import threading |
|
|
| import spaces |
|
|
| |
| |
| |
| import torch |
|
|
| |
| ON_ZEROGPU = os.getenv("SPACES_ZERO_GPU", "").lower() in ("1", "t", "true") |
|
|
| |
| GPU_DEVICE = "cuda" if ON_ZEROGPU else "cpu" |
|
|
| |
| |
| |
| _VECTORIZE_SECONDS = 120 |
|
|
|
|
| def cuda_state(tag: str) -> str: |
| """One-line snapshot of this process's CUDA state, for the Space logs. |
| |
| Reads the two fields that actually decide whether a ZeroGPU fork will |
| succeed, neither of which `spaces` patches (so both tell the truth): |
| |
| torch.cuda._initialized True once CUDA has REALLY been brought up here. |
| Must still be False in the parent at the moment |
| of the GPU call — otherwise the fork is doomed. |
| torch.cuda._is_in_bad_fork() |
| True when this process inherited an already- |
| initialised CUDA context across a fork. |
| |
| Also reports the thread, because "which thread" is the whole ballgame: the |
| spaces Torch modes only cover the main thread. |
| """ |
| try: |
| bad_fork = torch.cuda._is_in_bad_fork() |
| except Exception as e: |
| bad_fork = f"<{type(e).__name__}>" |
| current = threading.current_thread() |
| return ( |
| f"[cuda-state:{tag}] pid={os.getpid()}" |
| f" thread={current.name!r}" |
| f" is_main_thread={current is threading.main_thread()}" |
| f" torch.cuda._initialized={getattr(torch.cuda, '_initialized', '?')}" |
| f" _is_in_bad_fork={bad_fork}" |
| f" CUDA_VISIBLE_DEVICES={os.environ.get('CUDA_VISIBLE_DEVICES')!r}" |
| f" ON_ZEROGPU={ON_ZEROGPU} GPU_DEVICE={GPU_DEVICE}" |
| ) |
|
|
|
|
| @spaces.GPU(duration=_VECTORIZE_SECONDS) |
| def vectorize_on_gpu(pdf_path: str) -> None: |
| """Build and persist this paper's vectorstore with the embedder on GPU. |
| |
| Returns None deliberately. ZeroGPU runs this in its own GPU worker, so a |
| Chroma handle created here would carry a cuda-resident embedding model back |
| to a caller that no longer holds the GPU — useless at best, a crash at worst. |
| What crosses the boundary is the *persisted vectorstore on disk*, which is |
| device-independent. |
| |
| The caller then re-opens it on CPU, which costs nothing: build_vectorstore |
| short-circuits to a plain load as soon as the persist dir exists. |
| """ |
| |
| |
| |
| print(cuda_state("gpu-worker"), flush=True) |
|
|
| from vectorizeer import build_vectorstore |
|
|
| build_vectorstore(pdf_path, device=GPU_DEVICE) |
|
|