Spaces:
Running on Zero
Running on Zero
File size: 6,130 Bytes
434c049 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 | #!/usr/bin/env python3
"""
scripts/heartbeat_ping.py — Health monitor, warmup probe, and latency benchmark for HF Spaces.
Usage:
python3 scripts/heartbeat_ping.py --url https://abalanescu-flow.hf.space
python3 scripts/heartbeat_ping.py --url https://abalanescu-flow.hf.space --warmup
python3 scripts/heartbeat_ping.py --url https://abalanescu-flow.hf.space --chat --stream
"""
import os
import sys
import time
import json
import argparse
import urllib.request
import urllib.error
def make_request(url: str, headers: dict = None, data: dict = None, stream: bool = False):
headers = headers or {}
post_data = json.dumps(data).encode("utf-8") if data is not None else None
if data is not None and "Content-Type" not in headers:
headers["Content-Type"] = "application/json"
req = urllib.request.Request(url, data=post_data, headers=headers)
t0 = time.time()
try:
with urllib.request.urlopen(req, timeout=60) as resp:
elapsed = round((time.time() - t0) * 1000, 2)
if stream:
print(f"[HTTP {resp.status}] Streaming response ({elapsed}ms TTFB):")
full_text = []
for line in resp:
decoded = line.decode("utf-8").strip()
if not decoded:
continue
if decoded == "data: [DONE]":
break
if decoded.startswith("data: "):
try:
payload = json.loads(decoded[6:])
delta = payload.get("choices", [{}])[0].get("delta", {})
content = delta.get("content", "")
if content:
sys.stdout.write(content)
sys.stdout.flush()
full_text.append(content)
except Exception:
pass
sys.stdout.write("\n")
return {"status": resp.status, "latency_ms": elapsed, "streamed": "".join(full_text)}
else:
body = resp.read().decode("utf-8")
try:
parsed = json.loads(body)
except Exception:
parsed = body
return {"status": resp.status, "latency_ms": elapsed, "data": parsed}
except urllib.error.HTTPError as e:
elapsed = round((time.time() - t0) * 1000, 2)
err_body = e.read().decode("utf-8", errors="replace")
return {"status": e.code, "latency_ms": elapsed, "error": err_body}
except Exception as e:
elapsed = round((time.time() - t0) * 1000, 2)
return {"status": 0, "latency_ms": elapsed, "error": str(e)}
def main():
parser = argparse.ArgumentParser(description="ZeroGPU Space Heartbeat and Latency Probe")
parser.add_argument("--url", default="https://abalanescu-flow.hf.space", help="Base URL of HF Space")
parser.add_argument("--key", default=None, help="Bearer auth token (defaults to FLOW_API_KEY or HF_TOKEN)")
parser.add_argument("--warmup", action="store_true", help="Send 1-token warm-up probe")
parser.add_argument("--chat", action="store_true", help="Send a test chat completion prompt")
parser.add_argument("--prompt", default="Hello! Identify yourself and your model architecture in one short sentence.", help="Prompt for chat probe")
parser.add_argument("--stream", action="store_true", help="Test streaming response via SSE")
parser.add_argument("--model", default=None, help="Model ID override")
args = parser.parse_args()
base_url = args.url.rstrip("/")
token = args.key or os.environ.get("FLOW_API_KEY") or os.environ.get("HF_TOKEN")
headers = {"Authorization": f"Bearer {token}"} if token else {}
print(f"=== ZeroGPU Space Probe: {base_url} ===")
# 1. Health Probe
health_url = f"{base_url}/v1/health"
print(f"\n[1/3] Checking Health ({health_url})...")
health_res = make_request(health_url)
print(f"Status: HTTP {health_res['status']} in {health_res['latency_ms']}ms")
if "data" in health_res:
print(f"Data: {json.dumps(health_res['data'], indent=2)}")
# 2. Models List
models_url = f"{base_url}/v1/models"
print(f"\n[2/3] Querying Models ({models_url})...")
models_res = make_request(models_url, headers=headers)
print(f"Status: HTTP {models_res['status']} in {models_res['latency_ms']}ms")
models = []
if "data" in models_res and isinstance(models_res["data"], dict):
models = [m.get("id") for m in models_res["data"].get("data", [])]
print(f"Available Models ({len(models)}): {', '.join(models[:5])}")
# 3. Warmup or Chat
if args.warmup:
warmup_url = f"{base_url}/v1/warmup"
print(f"\n[3/3] Sending Warmup Probe ({warmup_url})...")
warm_res = make_request(warmup_url, headers=headers, data={})
print(f"Status: HTTP {warm_res['status']} in {warm_res['latency_ms']}ms")
if "data" in warm_res:
print(f"Warmup Result: {json.dumps(warm_res['data'], indent=2)}")
if args.chat:
chat_url = f"{base_url}/v1/chat/completions"
target_model = args.model or (models[0] if models else "default")
print(f"\n[3/3] Chat Completion Probe on '{target_model}' (stream={args.stream})...")
payload = {
"model": target_model,
"messages": [{"role": "user", "content": args.prompt}],
"temperature": 0.7,
"max_tokens": 128,
"stream": args.stream,
}
chat_res = make_request(chat_url, headers=headers, data=payload, stream=args.stream)
if not args.stream:
print(f"Status: HTTP {chat_res['status']} in {chat_res['latency_ms']}ms")
if "data" in chat_res:
print(f"Response:\n{json.dumps(chat_res['data'], indent=2)}")
elif "error" in chat_res:
print(f"Error: {chat_res['error']}")
print("\n=== Probe Finished ===")
if __name__ == "__main__":
main()
|