"""Run GPU work from custom HTML views through a real Gradio event. ZeroGPU charges each GPU call to the visitor who made it. It knows the visitor from the request, and it only treats a request as coming from the app (rather than an external API client) when it is a Gradio event fired by the page. `gr.HTML` server functions are neither, so a GPU call made from one is billed as an anonymous API call and quickly hits "You have exceeded your ZeroGPU runs limit", even on an Enterprise org's Space. So views ask for GPU work with `gpuCall(op, payload)` (below). That dispatches a window event; a hidden bridge component turns it into a Gradio "submit" event, the handler runs the op, and the result comes back through the bridge's value. Payload keys must not be named `path`: Gradio treats such dicts as file references. """ import re import gradio as gr TEMPLATE = '
' # Runs once, in the bridge component. SCRIPT = r""" window.addEventListener('df-gpu-request', e => trigger('submit', e.detail)); watch('value', () => { const v = props.value; if (v && v.nonce) window.dispatchEvent(new CustomEvent('df-gpu-result', { detail: v })); }); """ # Pasted into each view's script: `await gpuCall('op', {...})` resolves with the op's result. # `friendlyGpuError(msg)` turns a ZeroGPU quota or limit message into plain words (null for other errors). CLIENT = r""" function friendlyGpuError(msg) { const m = String(msg || ''); if (!/quota|runs limit|exceeded your/i.test(m)) return null; // not Playwright's 'Timeout 15000ms exceeded' const wait = (m.match(/try again in ([0-9:]+)/i) || [])[1]; return { title: 'Out of free GPU time for today', text: 'Every visitor gets a daily ZeroGPU quota on Hugging Face, and this demo has used yours up. Sign in to ' + 'Hugging Face for more, or come back ' + (wait ? 'in ' + wait : 'later') + '.', link: 'https://huggingface.co/login', detail: m }; } function gpuCall(op, payload, timeoutMs = 300000) { return new Promise((resolve, reject) => { const nonce = Math.random().toString(36).slice(2) + Date.now().toString(36); const done = (fn, arg) => { window.removeEventListener('df-gpu-result', on); clearTimeout(timer); fn(arg); }; const on = e => { if (!e.detail || e.detail.nonce !== nonce) return; const r = e.detail.result; if (r && r.__error) done(reject, new Error(r.__error)); else done(resolve, r); }; const timer = setTimeout(() => done(reject, new Error('The GPU request timed out. Please try again.')), timeoutMs); window.addEventListener('df-gpu-result', on); window.dispatchEvent(new CustomEvent('df-gpu-request', { detail: Object.assign({ op, nonce }, payload) })); }); } """ CSS = "#df-gpu-bridge { display: none !important; }" def handler(ops: dict): """The bridge's event handler: runs `ops[op](payload)` and reports errors as data.""" def run(evt: gr.EventData): data = evt._data if isinstance(getattr(evt, "_data", None), dict) else {} op, nonce = data.get("op"), data.get("nonce") try: if op not in ops: raise ValueError(f"unknown GPU op {op!r}") result = ops[op](data) except Exception as exc: # e.g. a ZeroGPU quota message, shown to the visitor by the view result = {"__error": str(exc) or type(exc).__name__} return {"nonce": nonce, "result": result} return run def friendly_error(message: str) -> str: """ZeroGPU quota and limit errors in plain words; other messages are returned unchanged.""" if not re.search(r"quota|runs limit|exceeded your", message, re.I): # not "Timeout 15000ms exceeded" return message wait = re.search(r"try again in ([0-9:]+)", message, re.I) return ("You have used up today's free GPU time on Hugging Face. Sign in to Hugging Face for more, or come back " + (f"in {wait.group(1)}" if wait else "later") + ".")