Spaces:
Running
Running
| """Uptime self-heal for the SWIA Commission Intake Space (pramodmisra/pif). | |
| Runs on a schedule (GitHub Actions cron, every ~15 min). Reads the Space's | |
| runtime stage and: | |
| * RUNNING / APP_STARTING / BUILDING / RUNNING_APP_STARTING / running-ish | |
| -> healthy or transient; do nothing (exit 0). | |
| * RUNTIME_ERROR / PAUSED / STOPPED / SLEEPING | |
| -> recoverable by a plain restart (bucket persists, DB safe per | |
| OPERATIONS.md 2/7); restart, then poll for recovery. | |
| * BUILD_ERROR / CONFIG_ERROR / NO_APP_FILE | |
| -> a code/config problem a restart won't fix; DO NOT restart, exit 1 | |
| so the Action fails and emails the owner. | |
| A plain restart is non-destructive: the persistent bucket pramodmisra/pif-db | |
| stays mounted at /data, so logins/submissions/CE edits survive. We never call | |
| set_space_volumes or any factory/reset action here. | |
| Secret: HF_TOKEN (write scope) from the environment - never hardcoded. | |
| """ | |
| from __future__ import annotations | |
| import os | |
| import sys | |
| import time | |
| from huggingface_hub import HfApi | |
| SPACE_REPO_ID = "pramodmisra/pif" | |
| # Stage buckets. Anything not listed is treated as "unknown" -> restart-and-see. | |
| HEALTHY = {"RUNNING", "APP_STARTING", "BUILDING", "RUNNING_APP_STARTING"} | |
| RECOVERABLE = {"RUNTIME_ERROR", "PAUSED", "STOPPED", "SLEEPING"} | |
| UNRECOVERABLE = {"BUILD_ERROR", "CONFIG_ERROR", "NO_APP_FILE"} | |
| RESTART_RETRIES = 3 # MAX attempts for the restart API call | |
| POLL_MAX = 20 # MAX post-restart status checks | |
| POLL_INTERVAL_S = 15 # seconds between checks (~5 min ceiling) | |
| def _stage(api: HfApi) -> str: | |
| """Current runtime stage as an upper-case string.""" | |
| return str(api.get_space_runtime(SPACE_REPO_ID).stage).upper() | |
| def _restart_with_backoff(api: HfApi) -> None: | |
| """Trigger a plain restart with exponential backoff; raise on total failure.""" | |
| last_err: Exception | None = None | |
| for attempt in range(RESTART_RETRIES): | |
| try: | |
| api.restart_space(SPACE_REPO_ID) | |
| return | |
| except Exception as err: # noqa: BLE001 - surface, don't swallow | |
| last_err = err | |
| wait = 2 ** attempt | |
| print(f" restart attempt {attempt + 1} failed: {err!r}; retry in {wait}s") | |
| time.sleep(wait) | |
| raise RuntimeError(f"restart_space failed after {RESTART_RETRIES} attempts") from last_err | |
| def _poll_until_running(api: HfApi) -> str: | |
| """Poll (bounded) until the Space leaves the starting states; return final stage.""" | |
| stage = _stage(api) | |
| for i in range(POLL_MAX): | |
| if stage == "RUNNING" or stage in UNRECOVERABLE: | |
| break | |
| time.sleep(POLL_INTERVAL_S) | |
| stage = _stage(api) | |
| print(f" [{i + 1}/{POLL_MAX}] stage = {stage}") | |
| return stage | |
| def main() -> None: | |
| token = os.environ.get("HF_TOKEN") | |
| if not token: | |
| print("ERROR: HF_TOKEN not set in environment.", file=sys.stderr) | |
| sys.exit(1) | |
| api = HfApi(token=token) | |
| stage = _stage(api) | |
| print(f"Space {SPACE_REPO_ID} stage = {stage}") | |
| if stage in HEALTHY: | |
| print("Healthy or transient; nothing to do.") | |
| return | |
| if stage in UNRECOVERABLE: | |
| print(f"ERROR: {stage} is a code/config failure a restart won't fix.", | |
| file=sys.stderr) | |
| sys.exit(1) | |
| # RECOVERABLE (or unknown) -> non-destructive restart, then confirm. | |
| print(f"{stage} is recoverable; issuing a plain restart (bucket persists)...") | |
| _restart_with_backoff(api) | |
| final = _poll_until_running(api) | |
| if final == "RUNNING": | |
| print("Recovered: Space is RUNNING.") | |
| return | |
| print(f"ERROR: Space did not recover; final stage = {final}.", file=sys.stderr) | |
| sys.exit(1) | |
| if __name__ == "__main__": | |
| main() | |