Spaces:
Sleeping
Sleeping
| """Pure runtime configuration and ZeroGPU-safe PEFT loading helpers.""" | |
| from __future__ import annotations | |
| import re | |
| import warnings | |
| from collections.abc import Mapping | |
| from collections.abc import Callable | |
| from dataclasses import dataclass | |
| from typing import Any | |
| DEFAULT_ADAPTER_REVISION = "3702c6cbc8d0bef2327e4c16d72d061247c2bbf2" | |
| DEFAULT_BASE_MODEL_ID = "Qwen/Qwen3-0.6B" | |
| DEFAULT_BASE_REVISION = "c1899de289a04d12100db370d81485cdf75e47ca" | |
| DEFAULT_GPU_SIZE = "large" | |
| DEFAULT_GPU_DURATION_SECONDS = 60 | |
| DEFAULT_MAX_INPUT_TOKENS = 512 | |
| DEFAULT_MAX_NEW_TOKENS = 256 | |
| _COMMIT_SHA_RE = re.compile(r"[0-9a-f]{40}") | |
| class AppSettings: | |
| """Validated settings needed by the model runtime.""" | |
| model_id: str | |
| model_revision: str | |
| base_model_id: str | |
| base_model_revision: str | |
| gpu_size: str | |
| gpu_duration_seconds: int | |
| max_input_tokens: int | |
| max_new_tokens: int | |
| def _text( | |
| environ: Mapping[str, str], | |
| name: str, | |
| default: str = "", | |
| ) -> str: | |
| return (environ.get(name) or default).strip() | |
| def _bounded_int( | |
| environ: Mapping[str, str], | |
| name: str, | |
| default: int, | |
| minimum: int, | |
| maximum: int, | |
| ) -> int: | |
| raw_value = _text(environ, name, str(default)) | |
| try: | |
| value = int(raw_value) | |
| except ValueError as exc: | |
| raise RuntimeError(f"{name} must be an integer.") from exc | |
| if not minimum <= value <= maximum: | |
| raise RuntimeError(f"{name} must be between {minimum} and {maximum}.") | |
| return value | |
| def load_app_settings(environ: Mapping[str, str]) -> AppSettings: | |
| """Resolve required values and apply safe defaults to every optional value.""" | |
| if not _text(environ, "HF_TOKEN"): | |
| raise RuntimeError( | |
| "HF_TOKEN is not set. Add it to .env or Space Settings > Secrets." | |
| ) | |
| model_id = _text(environ, "MODEL_ID") | |
| legacy_model_id = _text(environ, "MEDEL_ID") | |
| if not model_id and legacy_model_id: | |
| model_id = legacy_model_id | |
| warnings.warn( | |
| "MEDEL_ID is a supported compatibility alias; rename it to MODEL_ID.", | |
| RuntimeWarning, | |
| stacklevel=2, | |
| ) | |
| if not model_id: | |
| raise RuntimeError( | |
| "MODEL_ID is not set. Add it to .env or Space Settings > Variables." | |
| ) | |
| model_revision = _text( | |
| environ, | |
| "MODEL_REVISION", | |
| DEFAULT_ADAPTER_REVISION, | |
| ) | |
| base_model_id = _text(environ, "BASE_MODEL_ID", DEFAULT_BASE_MODEL_ID) | |
| base_model_revision = _text( | |
| environ, | |
| "BASE_MODEL_REVISION", | |
| DEFAULT_BASE_REVISION, | |
| ) | |
| for name, revision in ( | |
| ("MODEL_REVISION", model_revision), | |
| ("BASE_MODEL_REVISION", base_model_revision), | |
| ): | |
| if _COMMIT_SHA_RE.fullmatch(revision) is None: | |
| raise RuntimeError(f"{name} must be a pinned 40-character commit SHA.") | |
| gpu_size = _text(environ, "GPU_SIZE", DEFAULT_GPU_SIZE) | |
| if gpu_size not in {"large", "xlarge"}: | |
| raise RuntimeError("GPU_SIZE must be either 'large' or 'xlarge'.") | |
| return AppSettings( | |
| model_id=model_id, | |
| model_revision=model_revision, | |
| base_model_id=base_model_id, | |
| base_model_revision=base_model_revision, | |
| gpu_size=gpu_size, | |
| gpu_duration_seconds=_bounded_int( | |
| environ, | |
| "GPU_DURATION_SECONDS", | |
| DEFAULT_GPU_DURATION_SECONDS, | |
| 1, | |
| 300, | |
| ), | |
| max_input_tokens=_bounded_int( | |
| environ, | |
| "MAX_INPUT_TOKENS", | |
| DEFAULT_MAX_INPUT_TOKENS, | |
| 128, | |
| 1024, | |
| ), | |
| max_new_tokens=_bounded_int( | |
| environ, | |
| "MAX_NEW_TOKENS", | |
| DEFAULT_MAX_NEW_TOKENS, | |
| 32, | |
| 256, | |
| ), | |
| ) | |
| def load_peft_adapter_on_cpu( | |
| peft_model_type: Any, | |
| base_model: Any, | |
| *, | |
| model_id: str, | |
| model_revision: str, | |
| ) -> Any: | |
| """Deserialize adapter tensors on CPU before ZeroGPU's root CUDA transfer. | |
| ZeroGPU exposes CUDA emulation during module initialization. Without an | |
| explicit device, PEFT infers CUDA and asks safetensors to allocate on a real | |
| GPU before a GPU lease exists. Loading on CPU avoids that native CUDA path; | |
| the caller must then move the complete model to the selected runtime device. | |
| """ | |
| return peft_model_type.from_pretrained( | |
| base_model, | |
| model_id, | |
| revision=model_revision, | |
| is_trainable=False, | |
| torch_device="cpu", | |
| ) | |
| def require_zero_gpu_execution( | |
| environ: Mapping[str, str], | |
| *, | |
| server_pid: int, | |
| current_pid: int, | |
| cuda_available: bool, | |
| model_device_type: str, | |
| cuda_probe: Callable[[], None], | |
| ) -> bool: | |
| """Fail closed unless a hosted request is executing on a ZeroGPU worker. | |
| The ``spaces.GPU`` runtime invokes the decorated task in a forked worker | |
| after acquiring a real GPU. CUDA is emulated in the long-lived server | |
| process, so ``torch.cuda.is_available()`` alone is not sufficient proof. | |
| A distinct PID plus a successful CUDA operation verifies the execution | |
| boundary without preventing local CPU/MPS development. | |
| """ | |
| is_space = bool((environ.get("SPACE_ID") or "").strip()) | |
| is_zero_gpu = (environ.get("SPACES_ZERO_GPU") or "").lower() in { | |
| "1", | |
| "t", | |
| "true", | |
| } | |
| if not is_zero_gpu: | |
| if is_space: | |
| raise RuntimeError("This Space must run on ZeroGPU hardware.") | |
| return False | |
| if current_pid == server_pid: | |
| raise RuntimeError("Request did not enter a ZeroGPU GPU worker.") | |
| if not cuda_available or model_device_type != "cuda": | |
| raise RuntimeError("ZeroGPU worker does not expose the model on CUDA.") | |
| cuda_probe() | |
| return True | |