Spaces:
Running on Zero
Running on Zero
| #!/usr/bin/env python3 | |
| """ | |
| ratelimit.py — best-effort daily cap on the (paid) LLM transliteration stage. | |
| Detection is always free; only transliteration is metered. | |
| HONEST LIMITATIONS — this is cost control, not security: | |
| * State is in-memory. A Space restart or sleep resets every counter. | |
| * Identity is the client IP, which is shared behind NAT/VPN and rotatable. | |
| HF Spaces have no per-user identity available to the app. | |
| * GLOBAL_DAILY is the backstop that actually protects the API key: even if | |
| one caller rotates IPs freely, total spend per UTC day is bounded. | |
| Design note: check() does NOT consume quota. commit() is called only after a | |
| successful LLM call, so an OpenAI outage or a missing key never burns a user's | |
| three attempts. | |
| """ | |
| from __future__ import annotations | |
| import hashlib | |
| import os | |
| import threading | |
| from datetime import datetime, timezone | |
| PER_IP_DAILY = int(os.getenv('SPHINX_PER_IP_DAILY', '3')) | |
| GLOBAL_DAILY = int(os.getenv('SPHINX_GLOBAL_DAILY', '100')) | |
| _USAGE : dict[tuple[str, str], int] = {} # (ip_hash, utc_date) -> n | |
| _GLOBAL : dict[str, int] = {} # utc_date -> n | |
| _LOCK = threading.Lock() # Gradio serves requests threaded | |
| def _today() -> str: | |
| return datetime.now(timezone.utc).strftime('%Y-%m-%d') | |
| def client_id(request) -> str: | |
| """ | |
| Stable pseudonymous id for a caller. HF Spaces sit behind a proxy, so | |
| x-forwarded-for holds the real client; its first entry is the origin. | |
| Hashed — raw IPs are never stored. | |
| """ | |
| ip = 'unknown' | |
| if request is not None: | |
| headers = getattr(request, 'headers', None) or {} | |
| fwd = '' | |
| try: | |
| fwd = headers.get('x-forwarded-for', '') or '' | |
| except Exception: | |
| fwd = '' | |
| if fwd: | |
| ip = fwd.split(',')[0].strip() | |
| else: | |
| client = getattr(request, 'client', None) | |
| ip = getattr(client, 'host', None) or 'unknown' | |
| return hashlib.sha256(ip.encode('utf-8')).hexdigest()[:16] | |
| def remaining(request) -> int: | |
| """Attempts left today for this caller (ignores the global ceiling).""" | |
| key = (client_id(request), _today()) | |
| with _LOCK: | |
| return max(0, PER_IP_DAILY - _USAGE.get(key, 0)) | |
| def check(request) -> tuple[bool, int, str]: | |
| """ | |
| -> (allowed, remaining_after_a_hypothetical_use, message) | |
| Does not consume quota; call commit() on success. | |
| """ | |
| day = _today() | |
| key = (client_id(request), day) | |
| with _LOCK: | |
| used = _USAGE.get(key, 0) | |
| global_used = _GLOBAL.get(day, 0) | |
| if global_used >= GLOBAL_DAILY: | |
| return (False, 0, | |
| f'The shared daily transliteration budget for this Space ' | |
| f'({GLOBAL_DAILY}/day) is exhausted. Detection still works — ' | |
| f'try the LLM stage again tomorrow (UTC).') | |
| if used >= PER_IP_DAILY: | |
| return (False, 0, | |
| f'Daily limit reached: {PER_IP_DAILY} transliterations per day. ' | |
| f'Detection, reading order and cartouche matching still run — ' | |
| f'only the GPT stage is capped. Resets at 00:00 UTC.') | |
| return (True, PER_IP_DAILY - used - 1, '') | |
| def commit(request) -> int: | |
| """Consume one unit after a SUCCESSFUL call. -> remaining.""" | |
| day = _today() | |
| key = (client_id(request), day) | |
| with _LOCK: | |
| _USAGE[key] = _USAGE.get(key, 0) + 1 | |
| _GLOBAL[day] = _GLOBAL.get(day, 0) + 1 | |
| # bound memory: drop every day but today | |
| for k in [k for k in _USAGE if k[1] != day]: | |
| del _USAGE[k] | |
| for k in [k for k in _GLOBAL if k != day]: | |
| del _GLOBAL[k] | |
| return max(0, PER_IP_DAILY - _USAGE[key]) | |
| def _reset_for_tests() -> None: | |
| with _LOCK: | |
| _USAGE.clear() | |
| _GLOBAL.clear() | |